From 3168e603b41660846535eceeaad49e8d98e5ff36 Mon Sep 17 00:00:00 2001 From: Makisuo Date: Sat, 3 Oct 2026 00:19:42 +0200 Subject: [PATCH 1/2] Quote identifiers through the Dialect and gate dialect-only clauses Column refs, qualifiers, tables, aliases, CTE names and group/order keys render as Ident fragments the dialect quotes. ClickHouse writes them bare, so its output is unchanged. Untyped boolean and DateTime literals go through the dialect too. Dialect.clauses records which clauses exist: FORMAT fails to compile where it has no meaning, and a wrapped union gets the derived-table alias some databases require. Dialect.paramCodecs lets a dialect re-encode a portable param kind. --- src/ch/compile.ts | 44 ++++++++++++++++++++++++++------------ src/ch/dialect.test.ts | 38 +++++++++++++++++++++++++++++++++ src/ch/dialect.ts | 45 +++++++++++++++++++++++++++++---------- src/ch/expr.ts | 28 ++++++++++++++++++------ src/ch/literal.ts | 4 ++-- src/sql/literal-syntax.ts | 32 ---------------------------- src/sql/sql-fragment.ts | 34 +++++++++++++++++++++++------ src/sql/sql-syntax.ts | 38 +++++++++++++++++++++++++++++++++ 8 files changed, 190 insertions(+), 73 deletions(-) delete mode 100644 src/sql/literal-syntax.ts create mode 100644 src/sql/sql-syntax.ts diff --git a/src/ch/compile.ts b/src/ch/compile.ts index 4708634..9993619 100644 --- a/src/ch/compile.ts +++ b/src/ch/compile.ts @@ -12,7 +12,7 @@ import type { CHQuery, CHQueryState } from "./query" import type { CHUnionQuery } from "./union" import { createQualifiedColumnAccessor, createJoinedColumnAccessor, sourceAlias } from "./query" import { aliased, columnTypeOf } from "./expr" -import { raw, ident, compile as compileSqlFragment } from "../sql/sql-fragment" +import { raw, identPath, quoteIdent, quoteIdentPath, compile as compileSqlFragment } from "../sql/sql-fragment" import { splitTerminalClauses } from "../sql/terminal-clauses" import { compileQuery, type SqlQuery } from "../sql/sql-query" import { PARAM_MARKER_PREFIX, PARAM_PLACEHOLDER_PATTERN, paramSchema, type ParamKind } from "./param" @@ -65,9 +65,23 @@ const orderByClause = (specs: ReadonlyArray<[string, "asc" | "desc"]>): Array { + if (format === undefined) return undefined + const dialect = currentDialect() + if (!dialect.clauses.format) { + throw new QueryBuilderDefect({ + message: `CHQuery: format(${JSON.stringify(format)}) has no meaning for the ${dialect.name} dialect, which has no FORMAT clause`, + }) + } + return format +} + // CompiledQuery — bundles the SQL string with its output type so consumers // never need to cast manually. @@ -654,16 +668,17 @@ function compileInner< enclosingCtes: visibleCtes, }) fromSource = sourceOf(inner) - fromFragment = raw(`(${inner.sql}) AS ${state.fromQueryAlias}`) + fromFragment = raw(`(${inner.sql}) AS ${quoteIdent(state.fromQueryAlias ?? "")}`) } else if (state.fromUnion) { const inner = compileUnionInner(state.fromUnion, params, { deferParams, nested: true, enclosingCtes: visibleCtes }) fromSource = sourceOf(inner) - fromFragment = raw(`(\n${splitTerminalClauses(inner.sql).body}\n) AS ${state.fromQueryAlias}`) + const body = currentDialect().clauses.format ? splitTerminalClauses(inner.sql).body : inner.sql + fromFragment = raw(`(\n${body}\n) AS ${quoteIdent(state.fromQueryAlias ?? "")}`) } else { fromSource = sourceForTable(state.tableName, mainColumn) fromFragment = mainAlias !== state.tableName - ? raw(`${state.tableName} AS ${mainAlias}`) - : ident(state.tableName) + ? raw(`${quoteIdentPath(state.tableName)} AS ${quoteIdentPath(mainAlias)}`) + : identPath(state.tableName) } const sources: TenantSource[] = [fromSource] @@ -696,7 +711,7 @@ function compileInner< tableSql = `(${compiled.sql})` source = sourceOf(compiled) } else if (j.tableName) { - tableSql = j.tableName + tableSql = quoteIdentPath(j.tableName) source = sourceForTable( j.tableName, j.tenantColumn === undefined ? undefined : `${j.alias}.${j.tenantColumn}`, @@ -722,7 +737,7 @@ function compileInner< return { type: j.type, table: tableSql, - alias: j.alias, + alias: quoteIdent(j.alias), on: on ? compileSqlFragment(on.toFragment()) : undefined, } }) @@ -732,7 +747,7 @@ function compileInner< from: fromFragment, joins, where: whereFragments, - groupBy: state.groupByKeys.map((k) => raw(k)), + groupBy: state.groupByKeys.map((k) => raw(quoteIdent(k))), // Deliberately excluded from tenant evidence: by HAVING time the // rows are already aggregated, so the scan that produced them crossed // tenants no matter what this filters out. @@ -742,7 +757,7 @@ function compileInner< orderBy: orderByClause(state.orderBySpecs).map(raw), limit: state.limitValue != null ? raw(String(Math.round(state.limitValue))) : undefined, offset: state.offsetValue != null ? raw(String(Math.round(state.offsetValue))) : undefined, - format: options?.skipFormat ? undefined : state.formatValue, + format: options?.skipFormat ? undefined : formatClause(state.formatValue), } return compileQuery(sqlQuery) @@ -750,7 +765,7 @@ function compileInner< // Prepend CTE definitions if (resolvedCtes.length > 0) { - const cteDefs = resolvedCtes.map((c) => `${c.name} AS (\n${c.sql}\n)`).join(",\n") + const cteDefs = resolvedCtes.map((c) => `${quoteIdent(c.name)} AS (\n${c.sql}\n)`).join(",\n") sql = `WITH ${cteDefs}\n${sql}` } @@ -1067,7 +1082,7 @@ function compileUnionInner, Params extends Re state.outerOrderBySpecs.length > 0 || state.outerLimitValue != null || state.outerOffsetValue != null if (hasOuter) { - sql = `SELECT * FROM (\n${sql}\n)` + sql = `SELECT * FROM (\n${sql}\n)${currentDialect().clauses.derivedTableAlias ? ` AS ${quoteIdent("__union")}` : ""}` if (state.outerOrderBySpecs.length > 0) { sql += `\nORDER BY ${orderByClause(state.outerOrderBySpecs).join(", ")}` } @@ -1079,8 +1094,9 @@ function compileUnionInner, Params extends Re } } - if (state.formatValue) { - sql += `\nFORMAT ${state.formatValue}` + const format = formatClause(state.formatValue) + if (format) { + sql += `\nFORMAT ${format}` } let parameters: ReadonlyArray = [] diff --git a/src/ch/dialect.test.ts b/src/ch/dialect.test.ts index 842f59b..79ab8e4 100644 --- a/src/ch/dialect.test.ts +++ b/src/ch/dialect.test.ts @@ -19,6 +19,7 @@ const positional: Dialect = { // Standard SQL strings: a quote is doubled, a backslash is literal text. const standardQuote = (value: string) => `'${value.replace(/'/g, "''")}'` const literalDialect = (name: string, quoteString: (value: string) => string): Dialect => ({ + ...CH.clickhouseDialect, name, quoteString, literal: (value, context) => { @@ -182,3 +183,40 @@ describe("dialect literal syntax", () => { ) }) }) + +describe("dialect identifiers and clauses", () => { + const quoted: Dialect = { + ...CH.clickhouseDialect, + name: "quoted", + quoteIdent: (name) => `"${name.replace(/"/g, '""')}"`, + clauses: { format: false, derivedTableAlias: true }, + } + const services = CH.table("db.services", { OrgId: CH.string, Service: CH.string }, { tenantColumn: "OrgId" }) + + it("quotes columns, qualifiers, tables, aliases, group and order keys", () => { + const query = CH.from(events) + .innerJoin(services, "s", (main, s) => main.Service.eq(s.Service)) + .select(($) => ({ service: $.s.Service, count: CH.count() })) + .where(($) => [$.OrgId.eq("o")]) + .groupBy("service") + .orderBy(["count", "desc"]) + const { sql } = compileCHUnsafe(query, {}, { dialect: quoted }) + expect(sql).toContain(`"s"."Service" AS "service"`) + expect(sql).toContain(`INNER JOIN "db"."services" AS "s" ON "events"."Service" = "s"."Service"`) + expect(sql).toContain(`FROM "events"`) + expect(sql).toContain(`WHERE "events"."OrgId" = 'o'`) + expect(sql).toContain(`GROUP BY "service"`) + expect(sql).toContain(`ORDER BY "count" DESC`) + }) + + it("aliases a wrapped union and refuses FORMAT where the dialect has neither", () => { + const branch = CH.from(events) + .select(($) => ({ count: $.Count })) + .where(($) => [$.OrgId.eq("o")]) + const union = compileUnionUnsafe(CH.unionAll(branch, branch).orderBy(["count", "asc"]), {}, { dialect: quoted }) + expect(union.sql).toContain(`) AS "__union"\nORDER BY "count" ASC`) + + expect(() => compileCHUnsafe(branch.format("JSON"), {}, { dialect: quoted })).toThrow(/no FORMAT clause/) + expect(compileCHUnsafe(branch.format("JSON"), {}).sql).toMatch(/FORMAT JSON$/) + }) +}) diff --git a/src/ch/dialect.ts b/src/ch/dialect.ts index 28523d0..37c314a 100644 --- a/src/ch/dialect.ts +++ b/src/ch/dialect.ts @@ -2,16 +2,17 @@ // // What a compiled query needs to know about the database it is written for. // The builder, the tenant analysis, and row decoding are the same for every -// database; a dialect is the part that is not: how literals are written and how -// resolved param values reach the server. Identifier quoting, clause rendering, -// and column wire formats move behind it in later steps (see -// `design/dialects.md`). +// database; a dialect is the part that is not: how identifiers and literals are +// written, how resolved param values reach the server, which clauses exist, and +// how a portable param kind encodes. See `design/dialects.md`. +import type { Schema } from "effect" import { quoteClickHouseString } from "../sql/sql-fragment" -import { activeLiteralSyntax, withLiteralSyntax, type LiteralSyntax } from "../sql/literal-syntax" +import { activeSqlSyntax, withSqlSyntax, type SqlSyntax } from "../sql/sql-syntax" import { QueryBuilderError } from "./errors" import { sqlLiteral } from "./literal" import { PARAM_MARKER_PREFIX } from "./param" +import { chDateTimeLiteral } from "./types" /** * How resolved param values reach the server. @@ -38,6 +39,15 @@ export type ParamStyle = readonly reuse: boolean } +/** Clauses that exist in some dialects and not others. */ +export interface DialectClauses { + /** `FORMAT ` after a statement. A query with `.format()` fails to + * compile for a dialect without it. */ + readonly format: boolean + /** Whether a subquery in FROM needs an alias (`SELECT * FROM (…) AS u`). */ + readonly derivedTableAlias: boolean +} + /** * A database the builder writes SQL for. * @@ -46,36 +56,47 @@ export type ParamStyle = * finished text, so a value that spelled one would be rewritten too. A literal * that does is refused at compile time rather than trusted. */ -export interface Dialect extends LiteralSyntax { +export interface Dialect extends SqlSyntax { /** Shown in errors. */ readonly name: string readonly params: ParamStyle + readonly clauses: DialectClauses + /** + * The codec a portable param kind encodes with here, where it differs from + * the ClickHouse one: `param.bool` is `1`/`0` for ClickHouse and a boolean + * for Postgres, `param.dateTime` a zoneless string for one and an ISO-8601 + * instant for the other. Kinds not listed use the ClickHouse codec. + */ + readonly paramCodecs?: Readonly>> } /** ClickHouse, with params written into the SQL as literals. The default. */ export const clickhouseDialect: Dialect = { name: "clickhouse", + quoteIdent: (name) => name, quoteString: quoteClickHouseString, literal: sqlLiteral, + dateTimeLiteral: (value) => quoteClickHouseString(chDateTimeLiteral(value)), params: { _tag: "inline" }, + clauses: { format: true, derivedTableAlias: false }, } // The dialect of the enclosing compile, beside the syntax installed for the -// fragment renderer. Same save/restore discipline as `withLiteralSyntax`. +// fragment renderer. Same save/restore discipline as `withSqlSyntax`. let current: Dialect | undefined /** The dialect of the enclosing compile, or ClickHouse outside one. */ export const currentDialect = (): Dialect => current ?? clickhouseDialect -/** Run `body` with `dialect`'s literal syntax installed, checked as above. */ +/** Run `body` with `dialect`'s syntax installed, literals checked as above. */ export function withDialect(dialect: Dialect, body: () => A): A { // A nested compile for the dialect already installed keeps the checked // syntax that is there instead of wrapping it again. - if (current === dialect && activeLiteralSyntax() !== undefined) return body() + if (current === dialect && activeSqlSyntax() !== undefined) return body() const previous = current current = dialect try { - return withLiteralSyntax(checkedSyntax(dialect), body) + return withSqlSyntax(checkedSyntax(dialect), body) } finally { current = previous } @@ -86,9 +107,11 @@ export function withDialect(dialect: Dialect, body: () => A): A { export const checkedLiteral = (dialect: Dialect, value: unknown, context: string): string => checked(dialect, dialect.literal(value, context), context) -const checkedSyntax = (dialect: Dialect): LiteralSyntax => ({ +const checkedSyntax = (dialect: Dialect): SqlSyntax => ({ + quoteIdent: dialect.quoteIdent, quoteString: (value) => checked(dialect, dialect.quoteString(value), "a string literal"), literal: (value, context) => checkedLiteral(dialect, value, context), + dateTimeLiteral: (value) => checked(dialect, dialect.dateTimeLiteral(value), "a DateTime literal"), }) const checked = (dialect: Dialect, sql: string, context: string): string => { diff --git a/src/ch/expr.ts b/src/ch/expr.ts index 4e5099b..3f8e658 100644 --- a/src/ch/expr.ts +++ b/src/ch/expr.ts @@ -8,7 +8,8 @@ import { DateTime, Result, Schema } from "effect" import type { SqlFragment } from "../sql/sql-fragment" -import { raw, str, compile, as_ as sqlAs, lazy } from "../sql/sql-fragment" +import { raw, str, ident, compile, as_ as sqlAs, lazy } from "../sql/sql-fragment" +import { activeSqlSyntax } from "../sql/sql-syntax" import { chDateTimeLiteral, CHFloatResult, CHNumber, string as chString, type CHType, type InferTS } from "./types" import { encodeColumnLiteral } from "./literal" import { markTenantColumn, markTenantPredicate, tenantColumnOf, tenantPredicatesOf } from "./tenant" @@ -155,14 +156,24 @@ export function toFragment(value: unknown): SqlFragment { if (isExprLike(value)) return value.toFragment() if (typeof value === "string") return str(value) if (typeof value === "number") return raw(String(value)) - if (typeof value === "boolean") return raw(value ? "1" : "0") + if (typeof value === "boolean") return lazy(() => untypedLiteral(value)) // A DateTime column compares against a DateTime value, so the literal has to - // be ClickHouse's tz-less form rather than whatever `String(value)` produces. - if (DateTime.isDateTime(value)) return str(chDateTimeLiteral(DateTime.toUtc(value))) - if (value instanceof Date) return str(chDateTimeLiteral(DateTime.makeUnsafe(value))) + // be the dialect's own form (ClickHouse's is tz-less) rather than whatever + // `String(value)` produces. + if (DateTime.isDateTime(value)) return lazy(() => dateTimeLiteral(DateTime.toUtc(value))) + if (value instanceof Date) return lazy(() => dateTimeLiteral(DateTime.makeUnsafe(value))) return raw(String(value)) } +/** A value with no column type to encode it, in the active dialect's syntax. + * ClickHouse writes booleans as `1`/`0`, which is also the rendering outside + * a compile. */ +const untypedLiteral = (value: boolean): string => + activeSqlSyntax()?.literal(value, "an untyped boolean") ?? (value ? "1" : "0") + +const dateTimeLiteral = (value: DateTime.Utc): string => + activeSqlSyntax()?.dateTimeLiteral(value) ?? compile(str(chDateTimeLiteral(value))) + // Expr implementation /** Whether a codec accepts `null` — asked, not inferred from its AST, so it @@ -312,7 +323,10 @@ export function makeColumnRef { - const fragment = raw(name) + // `alias.Column` when qualified: the qualifier is quoted segment by segment, + // the column as one identifier (a ClickHouse `Nested` column has a dot). + const qualified = columnName !== undefined && name.endsWith(`.${columnName}`) + const fragment = qualified ? ident(columnName, name.slice(0, -columnName.length - 1)) : ident(name) const base = makeExpr>( fragment, columnType?.schema as Schema.Codec, any> | undefined, @@ -366,7 +380,7 @@ export function makeColumnRef { - return makeExpr(lazy(() => `${name}[${compile(str(key))}]`), columnType?.element?.schema) + return makeExpr(lazy(() => `${compile(fragment)}[${compile(str(key))}]`), columnType?.element?.schema) }, }, ) as ColumnRef diff --git a/src/ch/literal.ts b/src/ch/literal.ts index 09a5612..8ec6dbf 100644 --- a/src/ch/literal.ts +++ b/src/ch/literal.ts @@ -14,7 +14,7 @@ import { Result, Schema } from "effect" import { QueryBuilderError } from "./errors" import { quoteClickHouseString } from "../sql/sql-fragment" -import { activeLiteralSyntax } from "../sql/literal-syntax" +import { activeSqlSyntax } from "../sql/sql-syntax" import type { CHType } from "./types" const isPlainObject = (value: unknown): value is Record => @@ -89,7 +89,7 @@ export function sqlLiteral(value: unknown, context: string): string { * fails here — while building the SQL — instead of becoming part of it. */ export function encodeLiteral(schema: Schema.Codec, value: unknown, context: string): string { - return (activeLiteralSyntax()?.literal ?? sqlLiteral)(encodeValue(schema, value, context), context) + return (activeSqlSyntax()?.literal ?? sqlLiteral)(encodeValue(schema, value, context), context) } /** diff --git a/src/sql/literal-syntax.ts b/src/sql/literal-syntax.ts deleted file mode 100644 index c714a2c..0000000 --- a/src/sql/literal-syntax.ts +++ /dev/null @@ -1,32 +0,0 @@ -// Literal syntax of the dialect being compiled for. -// -// Literals are written while a query's callbacks run, deep inside expression -// code that has no dialect argument to read. `compile` installs the dialect's -// syntax here for the duration of the compile instead, the same way -// `withSubqueryCompiler` hands down the subquery renderer. A leaf module so the -// fragment renderer can read it without importing the builder. - -export interface LiteralSyntax { - /** A string as a quoted literal. */ - readonly quoteString: (value: string) => string - /** An encoded wire value (string, number, boolean, null, arrays and records - * of those) as a literal. Throws for a value it cannot write. */ - readonly literal: (value: unknown, context: string) => string -} - -// Compilation is synchronous. Save/restore makes nested compilations -// independent, including when a callback throws. -let current: LiteralSyntax | undefined - -export function withLiteralSyntax(syntax: LiteralSyntax, body: () => A): A { - const previous = current - current = syntax - try { - return body() - } finally { - current = previous - } -} - -/** The syntax installed by the enclosing compile, if there is one. */ -export const activeLiteralSyntax = (): LiteralSyntax | undefined => current diff --git a/src/sql/sql-fragment.ts b/src/sql/sql-fragment.ts index 366e8ce..3217776 100644 --- a/src/sql/sql-fragment.ts +++ b/src/sql/sql-fragment.ts @@ -1,5 +1,5 @@ import { Data } from "effect" -import { activeLiteralSyntax } from "./literal-syntax" +import { activeSqlSyntax } from "./sql-syntax" // ClickHouse string escaping @@ -33,7 +33,14 @@ export const quoteClickHouseString = (value: string): string => `'${escapeClickH /** A string literal in the syntax of the dialect being compiled for, or * ClickHouse's outside a compile. */ -const quoteString = (value: string): string => (activeLiteralSyntax()?.quoteString ?? quoteClickHouseString)(value) +const quoteString = (value: string): string => (activeSqlSyntax()?.quoteString ?? quoteClickHouseString)(value) + +/** One identifier in the syntax of the dialect being compiled for. ClickHouse + * names are written bare, which is also the rendering outside a compile. */ +export const quoteIdent = (name: string): string => activeSqlSyntax()?.quoteIdent(name) ?? name + +/** A dotted path (`db.table`, `alias.Column`) with each segment quoted. */ +export const quoteIdentPath = (path: string): string => path.split(".").map(quoteIdent).join(".") // SQL Fragment AST @@ -44,8 +51,14 @@ export type SqlFragment = Data.TaggedEnum<{ Str: { readonly value: string } /** Integer parameter: produces the number as string, rounded */ Int: { readonly value: number } - /** Column or table identifier (unquoted — ClickHouse style) */ - Ident: { readonly name: string } + /** + * An identifier, quoted by the dialect being compiled for. + * + * `name` is one identifier and is never split, so a ClickHouse `Nested` + * column such as `Events.Name` stays whole. `qualifier` is the dotted path in + * front of it (`alias`, `db.table`), quoted segment by segment. + */ + Ident: { readonly name: string; readonly qualifier?: string } /** A list of fragments joined by a separator (empty strings from When(false) are filtered) */ Join: { readonly separator: string; readonly fragments: ReadonlyArray } /** An aliased expression: AS */ @@ -72,7 +85,13 @@ const Frag = Data.taggedEnum() export const raw = (sql: string): SqlFragment => Frag.Raw({ sql }) export const str = (value: string): SqlFragment => Frag.Str({ value }) export const int = (value: number): SqlFragment => Frag.Int({ value }) -export const ident = (name: string): SqlFragment => Frag.Ident({ name }) +export const ident = (name: string, qualifier?: string): SqlFragment => + Frag.Ident(qualifier === undefined ? { name } : { name, qualifier }) +/** A dotted path such as `db.table`, split into qualifier and name. */ +export const identPath = (path: string): SqlFragment => { + const dot = path.lastIndexOf(".") + return dot === -1 ? ident(path) : ident(path.slice(dot + 1), path.slice(0, dot)) +} export const join = (separator: string, ...fragments: ReadonlyArray): SqlFragment => Frag.Join({ separator, fragments }) export const as_ = (expr: SqlFragment, alias: string): SqlFragment => Frag.As({ expr, alias }) @@ -86,9 +105,10 @@ export const compile: (fragment: SqlFragment) => string = Frag.$match({ Raw: ({ sql }) => sql, Str: ({ value }) => quoteString(value), Int: ({ value }) => String(Math.round(value)), - Ident: ({ name }) => name, + Ident: ({ name, qualifier }) => + qualifier === undefined ? quoteIdent(name) : `${quoteIdentPath(qualifier)}.${quoteIdent(name)}`, Join: ({ separator, fragments }) => fragments.map(compile).filter(Boolean).join(separator), - As: ({ expr, alias }) => `${compile(expr)} AS ${alias}`, + As: ({ expr, alias }) => `${compile(expr)} AS ${quoteIdent(alias)}`, When: ({ condition, fragment }) => (condition ? compile(fragment) : ""), Lazy: ({ render }) => render(), }) diff --git a/src/sql/sql-syntax.ts b/src/sql/sql-syntax.ts new file mode 100644 index 0000000..a4e6af0 --- /dev/null +++ b/src/sql/sql-syntax.ts @@ -0,0 +1,38 @@ +// Syntax of the dialect being compiled for. +// +// Identifiers and literals are written while a query's callbacks run, deep +// inside expression code that has no dialect argument to read. `compile` +// installs the dialect's syntax here for the duration of the compile instead, +// the same way `withSubqueryCompiler` hands down the subquery renderer. A leaf +// module so the fragment renderer can read it without importing the builder. + +import type { DateTime } from "effect" + +export interface SqlSyntax { + /** One identifier (a column, alias, table or schema name), quoted as needed. */ + readonly quoteIdent: (name: string) => string + /** A string as a quoted literal. */ + readonly quoteString: (value: string) => string + /** An encoded wire value (string, number, boolean, null, arrays and records + * of those) as a literal. Throws for a value it cannot write. */ + readonly literal: (value: unknown, context: string) => string + /** A point in time compared against an expression with no declared type. */ + readonly dateTimeLiteral: (value: DateTime.Utc) => string +} + +// Compilation is synchronous. Save/restore makes nested compilations +// independent, including when a callback throws. +let current: SqlSyntax | undefined + +export function withSqlSyntax(syntax: SqlSyntax, body: () => A): A { + const previous = current + current = syntax + try { + return body() + } finally { + current = previous + } +} + +/** The syntax installed by the enclosing compile, if there is one. */ +export const activeSqlSyntax = (): SqlSyntax | undefined => current From 2abbdd47cdf84832c867ec95e50ac4dcb9cdc38c Mon Sep 17 00:00:00 2001 From: Makisuo Date: Sat, 3 Oct 2026 00:29:41 +0200 Subject: [PATCH 2/2] Add a Postgres dialect New ./postgres entry point: postgresDialect (double-quoted identifiers, standard string literals, $n binding), Postgres column types whose codecs accept what common drivers send, a function catalog for what Postgres spells differently (count(*), FILTER (WHERE ...), percentile_cont, date_trunc in UTC, date_bin, array_agg, ->>), and a compile that defaults to Postgres. Core changes it needed: Dialect.paramCodecs now re-encodes portable param kinds (param.bool binds a boolean, param.dateTime an ISO instant), and Dialect.clauses.groupByAlias writes GROUP BY keys by position where a bare name would resolve to an input column first. Tests run every query on PGlite, so they pass only if Postgres accepts the SQL and the rows decode. ClickHouse output is unchanged. --- CHANGELOG.md | 13 ++ README.md | 2 + bun.lock | 3 + design/dialects.md | 61 ++++++---- docs/README.md | 1 + docs/params-and-compilation.md | 35 ++++-- docs/postgres.md | 120 ++++++++++++++++++ docs/reference.md | 9 +- package.json | 5 + scripts/check-doc-examples.mjs | 7 ++ src/ch/compile.ts | 19 ++- src/ch/dialect.test.ts | 2 +- src/ch/dialect.ts | 5 +- src/ch/index.ts | 2 +- src/pg/dialect.ts | 86 +++++++++++++ src/pg/functions.ts | 120 ++++++++++++++++++ src/pg/postgres.test.ts | 216 +++++++++++++++++++++++++++++++++ src/pg/types.ts | 156 ++++++++++++++++++++++++ src/postgres.ts | 34 ++++++ tsdown.config.ts | 1 + 20 files changed, 855 insertions(+), 42 deletions(-) create mode 100644 docs/postgres.md create mode 100644 src/pg/dialect.ts create mode 100644 src/pg/functions.ts create mode 100644 src/pg/postgres.test.ts create mode 100644 src/pg/types.ts create mode 100644 src/postgres.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 878d3c7..a54a655 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,18 @@ # Changelog +## Unreleased + +- Add `Dialect`: how a compiled query writes identifiers and literals, binds params, and + which clauses exist. `compile` and friends take `options.dialect`; `clickhouseDialect` is the + default and its output is unchanged. +- Add `CompiledQuery.parameters`: the values a binding dialect sends beside `sql`, empty for + ClickHouse. Code that builds a `CompiledQuery` by hand must now supply it. +- Add the `./postgres` entry point: `postgresDialect` (quoted identifiers, `$n` binding), + Postgres column types and functions, and a `compile` that defaults to Postgres. +- A literal that would contain the param marker `__PARAM_` now fails the compile with + `InvalidLiteral` instead of relying on each dialect's escaping. +- `compileUnionUnsafe` no longer accepts the internal `enclosingCtes` option. + ## 0.2.0 - Require Effect `^4.0.0`. Effect 4.0.0 moved `effect/unstable/*` to `effect/*` diff --git a/README.md b/README.md index 5d604c8..48d7234 100644 --- a/README.md +++ b/README.md @@ -151,6 +151,7 @@ Full guides live in [`docs/`](./docs/README.md): | [Decoding results](./docs/decoding-results.md) | `rowSchema`, `decodeRows`, decode errors | | [Running a query](./docs/running-queries.md) | Executing the SQL with a real client, wire settings, `SETTINGS` | | [Tenant scoping](./docs/tenant-scoping.md) | `tenantColumn`, what marks a query scoped, `crossTenant()` | +| [Postgres](./docs/postgres.md) | The Postgres dialect, its column types and functions | | [Extending the DSL](./docs/extending.md) | `defineFn`, raw escape hatches, handwritten SQL | | [API reference](./docs/reference.md) | Full export catalog by module, plus error types | @@ -166,6 +167,7 @@ regressions live in [`src/docs-examples.test.ts`](./src/docs-examples.test.ts). | `@maple-dev/effect-clickhouse/types` | Column-type constructors (`string`, `uint64`, `dateTime`, `map`, `array`, `nullable`, …) and the `CH*` type descriptors. | | `@maple-dev/effect-clickhouse/expr` | Kitchen-sink namespace: every expression helper plus all ClickHouse functions under their raw names (`min_`, `toString_`, `toStartOfInterval`, `dynamicColumn`, …). Handy for `import * as CH`. | | `@maple-dev/effect-clickhouse/sql` | The low-level `SqlFragment` AST (`raw`, `ident`, `compile`, …) for hand-rolling fragments. | +| `@maple-dev/effect-clickhouse/postgres` | The Postgres dialect: `postgresDialect`, Postgres column types and functions, and a `compile` that defaults to Postgres. | ## Extending with custom functions diff --git a/bun.lock b/bun.lock index e42b1b9..8cc4b6c 100644 --- a/bun.lock +++ b/bun.lock @@ -10,6 +10,7 @@ "@effect/platform-bun": "4.0.0", "@effect/sql-clickhouse": "4.0.0", "@effect/vitest": "4.0.0", + "@electric-sql/pglite": "0.3.15", "@types/node": "^22.10.2", "effect": "4.0.0", "expect-type": "^1.3.0", @@ -42,6 +43,8 @@ "@effect/vitest": ["@effect/vitest@4.0.0", "", { "peerDependencies": { "effect": "^4.0.0", "vitest": ">=5.0.0 <6.0.0" } }, "sha512-kGElCPFGtP+CgMCwV/HjJWqtE3zYo6I1+8MSA54l+EI24lPU9p2Efa+TBp1jfkg9j711KKzEqgSAy3VzTsb76Q=="], + "@electric-sql/pglite": ["@electric-sql/pglite@0.3.15", "", {}, "sha512-Cj++n1Mekf9ETfdc16TlDi+cDDQF0W7EcbyRHYOAeZdsAe8M/FJg18itDTSwyHfar2WIezawM9o0EKaRGVKygQ=="], + "@jridgewell/sourcemap-codec": ["@jridgewell/sourcemap-codec@1.6.0", "", {}, "sha512-T7jf+5zgsZHwNJ4lvQ7/aezbyk0nNX+zJVWpmHA7VYsEx7a7qr5Rg5IbtJFqkgze5Y2sruq1RUY8Q837Od7iFw=="], "@oxc-project/types": ["@oxc-project/types@0.148.0", "", {}, "sha512-Nm4s/jB+4FpFsPhWGEC4h7rzksesmtnMXomo6rCMcg/b8zLQuOziRgkCS1fxDCXOlJB/6Q8oABOZ/OP6RIPj9A=="], diff --git a/design/dialects.md b/design/dialects.md index f1ddc1c..055f2ee 100644 --- a/design/dialects.md +++ b/design/dialects.md @@ -1,6 +1,8 @@ # Dialects: one builder, several databases -Status: steps 1 and 2 landed (params and literals go through a `Dialect`). Steps 3 to 6 are open. +Status: done. ClickHouse and Postgres both compile from the same builder; see +[`docs/postgres.md`](../docs/postgres.md). Steps 4 and 5 landed differently from the plan, as +noted below. ## Goal @@ -15,14 +17,14 @@ Both were read from source (Kysely 0.28.17, Drizzle 1.0.0-rc.5). | Idea | Where it comes from | How it lands here | | --- | --- | --- | -| Builders produce a config object; a dialect turns it into SQL | Drizzle `PgDialect.buildSelectQuery(config)` | `CHQueryState` already is that config. `compile.ts` becomes the ClickHouse dialect's `buildSelect` | -| SQL as a chunk tree rendered at the end with `escapeName` / `escapeParam` / `escapeString` | Drizzle `SQL` chunks + `BuildQueryConfig` | `SqlFragment` gets a real `Param` chunk; `Str`/`Lazy` stop rendering ClickHouse text early | +| Builders produce a config object; a dialect turns it into SQL | Drizzle `PgDialect.buildSelectQuery(config)` | `CHQueryState` already is that config; `compile.ts` reads the dialect where clauses differ (step 5) | +| SQL as a chunk tree rendered at the end with `escapeName` / `escapeParam` / `escapeString` | Drizzle `SQL` chunks + `BuildQueryConfig` | `Str` and `Ident` render through the installed dialect; params stay text placeholders resolved once per statement | | Inline vs bound params as a switch | Drizzle `inlineParams` | `ParamStyle`: `inline` (ClickHouse today) or `bind` | | A small override surface per dialect | Kysely `DefaultQueryCompiler` hooks (MySQL overrides ~10 methods) | The `Dialect` interface grows hook by hook, never a copy of the compiler | -| Capability flags instead of dialect checks | Kysely `DialectAdapter` (`supportsReturning`, ...) | Flags such as `supportsFilterClause`, `aliasInWhere`, `limitBy` | +| Capability flags instead of dialect checks | Kysely `DialectAdapter` (`supportsReturning`, ...) | `Dialect.clauses`: `format`, `derivedTableAlias`, `groupByAlias` | | A compiled query that keeps its structure | Kysely `CompiledQuery.query` | `CompiledQuery` already carries tenant scope and row schema; `parameters` added in step 1 | -| Rewrites as passes over the tree | Kysely plugins (`transformQuery` / `transformResult`) | Tenant-scope proof and empty-`IN` handling as passes over the state | -| Logical type separate from how a driver sends it | Drizzle v1 codecs, `refineGenericPgCodecs` per driver | `CHType` splits into a type and a transport codec (step 3) | +| Rewrites as passes over the tree | Kysely plugins (`transformQuery` / `transformResult`) | Not adopted yet; tenant scope is still derived inside `compile` | +| Logical type separate from how a driver sends it | Drizzle v1 codecs, `refineGenericPgCodecs` per driver | Per-dialect type sets on the shared descriptor, codecs tolerant of every common driver (step 4); `Dialect.paramCodecs` for params | | Shared operators, per-dialect function catalogs | Drizzle `sql/expressions` vs `pg-core` / `mysql-core` | `eq`, `and`, `in_` in core; `countIf`, `percentileCont` per dialect | What we deliberately do not copy: @@ -47,23 +49,40 @@ What we deliberately do not copy: have no dialect argument. The `__PARAM_` safety is now enforced rather than assumed: a literal that contains the marker fails with `InvalidLiteral`, so a dialect with weaker escaping cannot turn a value into a placeholder. -3. **Identifier quoting.** Postgres folds unquoted names to lower case, so `OrgId` must be - written `"OrgId"`. Column refs, aliases, qualified names, `groupBy` keys and `orderBy` - specs are raw text today (`raw(name)` in `expr.ts`, `orderByClause` in `compile.ts`), so - this is its own step: column refs become `Ident` fragments that carry their qualifier. -4. **Split `CHType`.** A logical type (`sql` name, TS type) plus a codec for the transport. The - UInt64-as-string rule belongs to ClickHouse's `FORMAT JSON` over HTTP, not to ClickHouse; - the native client sends something else, and Postgres drivers send int8 as a string and - timestamptz as a `Date`. -5. **Move `compile.ts` into `ClickHouseDialect.buildSelect(state)`.** `compileQuery` and the - terminal clauses (`FORMAT`, `SETTINGS`, `LIMIT BY`) become ClickHouse-only. -6. **Postgres dialect.** `pg.T` column types, `$n` binding, `"ident"` quoting, a function - catalog covering the common cases (`FILTER (WHERE ...)`, `date_bin`, `percentile_cont`, - jsonb access), and capability flags for what ClickHouse allows and Postgres does not - (select aliases in `WHERE`/`HAVING`, default values instead of `NULL` in outer joins). +3. **Identifier quoting (done).** Column refs are `Ident` fragments carrying their qualifier + (`ident(column, "alias")`), so a ClickHouse `Nested` column such as `Events.Name` is never + split while `db.table` qualifiers are quoted per segment. Tables, aliases, CTE names, and + group/order keys go through `Dialect.quoteIdent` too. Untyped boolean and DateTime + literals moved behind the dialect in the same step (`dateTimeLiteral`). +4. **Column types (done, differently).** The plan was to split `CHType` into a logical type + plus a transport codec. Postgres turned out not to need the split: its types reuse the + `CHType` descriptor (`PgType` is an alias) with codecs that accept every wire form a common + driver sends (int8 as number, string or `bigint`; timestamptz as `Date` or text), the same + approach `CHNumber` already takes for quoted 64-bit integers. Portable param kinds that + must encode differently (`param.bool`, `param.dateTime`) are re-encoded per dialect by + `Dialect.paramCodecs`. A per-driver codec axis can come later if a consumer needs one. +5. **Clauses (done, differently).** The plan was to move `compile.ts` into a + `ClickHouseDialect.buildSelect(state)`. The clauses Postgres needed to differ were few: + no `FORMAT`, an alias on a wrapped union, and GROUP BY by position (Postgres reads a bare + name as an input column before a select alias). Those are `Dialect.clauses` flags read at + three places in `compile.ts`, which kept ClickHouse output byte-identical without + relocating ~1100 lines. A dialect that needs a structurally different statement is the + point to revisit this. +6. **Postgres dialect (done).** `@maple-dev/effect-clickhouse/postgres`: `postgresDialect` + (double-quoted identifiers, standard strings with `E'…'` only to escape the param marker, + `$n` binding), column types, a function catalog (`count(*)`, `FILTER (WHERE …)`, + `percentile_cont`, `date_trunc(…, 'UTC')`, `date_bin`, `array_agg`, `->>`), and a + `compile` that defaults to Postgres. `src/pg/postgres.test.ts` runs every query on PGlite + (Postgres 17 in WASM), so a test passes only if Postgres accepts the SQL and the rows decode. ## Open questions - ClickHouse also supports server-side binding (`{name:Type}` with `query_params`). Adding it needs the param kind to name a ClickHouse type, which `param.of` already has. -- Whether the package splits (`core`, `clickhouse`, `postgres` entry points) at step 4 or 5. +- The package is still named for ClickHouse and the root entry still exports the ClickHouse + function catalog. Renaming, or a `core` entry without ClickHouse functions, is a packaging + decision for a later release. +- The ClickHouse function catalog writes ClickHouse SQL whatever the dialect. Functions could + refuse to compile under another dialect instead of producing SQL that fails at the server. +- MySQL or SQLite would need `?` placeholders (`reuse: false` already exists) and their own + types and functions; nothing in the core should need to change. diff --git a/docs/README.md b/docs/README.md index 74b1f50..1589c5e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -52,6 +52,7 @@ Roughly in reading order. | [Agent benchmark playbook](./benchmark-agent.md) | Repeatable optimization workflow and evidence checklist | | [Tenant scoping](./tenant-scoping.md) | `tenantScope`, what marks a query scoped, `crossTenant()` | | [Extending the DSL](./extending.md) | `defineFn`, raw escape hatches, handwritten SQL | +| [Postgres](./postgres.md) | The Postgres dialect, its column types and functions | ## Reference diff --git a/docs/params-and-compilation.md b/docs/params-and-compilation.md index 92c1c7b..cdfb8b2 100644 --- a/docs/params-and-compilation.md +++ b/docs/params-and-compilation.md @@ -189,13 +189,16 @@ Use `decodeRows` to validate wire values against the row schema. ## Dialects -A `Dialect` decides how literals are written and how resolved params reach the server. Its -`quoteString` and `literal` write every string fragment, every value compared against a column, -and every inline param, for the length of the compile. The default, `clickhouseDialect`, -writes each value into the SQL as a ClickHouse literal and leaves -`parameters` empty. A dialect whose `params` style is `bind` leaves a placeholder instead and -returns the encoded values in `parameters`, numbered once across the whole statement, unions -and subqueries included: +A `Dialect` is the database a query is compiled for: how identifiers and literals are written, +how resolved params reach the server, and which clauses exist. It is installed for the length of +the compile, so every column reference, string fragment, compared value and inline param goes +through it. The default, `clickhouseDialect`, writes names bare and each param value into the SQL +as a ClickHouse literal, and leaves `parameters` empty. `postgresDialect`, from the `/postgres` +entry point, is the other built-in one; see [Postgres](./postgres.md). + +A dialect whose `params` style is `bind` leaves a placeholder instead and returns the encoded +values in `parameters`, numbered once across the whole statement, unions and subqueries +included: ```ts const numbered: CH.Dialect = { @@ -212,15 +215,25 @@ const compiled = CH.compileUnsafe(query, { orgId: "org_1" }, { dialect: numbered `reuse: true` lets one numbered placeholder stand for every use of a param; set it to `false` for positional `?` placeholders, which bind a value each time they appear. Either way a bound value is the column codec's wire form, the same value an inline literal is written from, and a -missing or ill-typed param still fails the compile. +missing or ill-typed param still fails the compile. `paramCodecs` re-encodes a portable param +kind where the database wants another form: Postgres binds `param.bool` as a boolean rather +than `1`/`0`. + +| Member | Purpose | +| --- | --- | +| `quoteIdent` | One identifier: a column, alias, table or schema name | +| `quoteString`, `literal` | A string, or any encoded wire value, as a literal | +| `dateTimeLiteral` | A point in time compared against an expression with no declared type | +| `params` | `inline`, or `bind` with a placeholder function | +| `clauses.format` | Whether `FORMAT` exists; `.format()` fails to compile where it does not | +| `clauses.derivedTableAlias` | Whether a subquery in FROM needs an alias | +| `clauses.groupByAlias` | Whether GROUP BY resolves select aliases; if not, keys are written by position | +| `paramCodecs` | Per-kind codec overrides for `param.*` | Params are resolved by rewriting placeholders in the finished SQL, so a dialect's literals must never spell the param marker `__PARAM_`. ClickHouse writes it as `\x5F_PARAM_`; a literal that does contain the marker fails the compile with `InvalidLiteral` rather than being rewritten. -Identifiers, clauses, and functions are still ClickHouse SQL; a dialect changes literals and -param binding today. - ## Handwritten SQL When you need SQL the builder cannot express, `rawCompiledQuery` wraps a string in the same diff --git a/docs/postgres.md b/docs/postgres.md new file mode 100644 index 0000000..bf208b5 --- /dev/null +++ b/docs/postgres.md @@ -0,0 +1,120 @@ +# Postgres + +The same builder writes Postgres SQL. Queries, params, tenant scoping and row decoding work +as they do for ClickHouse; the `/postgres` entry point supplies the parts that differ: column +types whose codecs read what Postgres drivers send, functions spelled the Postgres way, and a +`compile` that defaults to the Postgres dialect. + +```ts title="postgres-quickstart.ts" +import { PGlite } from "@electric-sql/pglite" +import { Effect } from "effect" +import * as CH from "@maple-dev/effect-clickhouse" +import * as PG from "@maple-dev/effect-clickhouse/postgres" + +const Requests = CH.table( + "requests", + { OrgId: PG.text, Route: PG.text, DurationMs: PG.int8, At: PG.timestamptz }, + { tenantColumn: "OrgId" }, +) + +const query = CH.from(Requests) + .select(($) => ({ + route: $.Route, + count: PG.count(), + slow: PG.countIf($.DurationMs.gte(500)), + p50: PG.percentileCont(0.5, $.DurationMs), + })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId")), $.At.gte(CH.param.dateTime("since"))]) + .groupBy("route") + .orderBy(["count", "desc"]) + +export const compiled = PG.compileUnsafe(query, { orgId: "org_1", since: new Date("2026-01-01T00:00:00Z") }) +// compiled.sql: ... WHERE "requests"."OrgId" = $1 AND "requests"."At" >= $2 ... +// compiled.parameters: ["org_1", "2026-01-01T00:00:00.000Z"] + +const db = new PGlite() +await db.exec(` + CREATE TABLE requests ("OrgId" text, "Route" text, "DurationMs" int8, "At" timestamptz); + INSERT INTO requests VALUES + ('org_1', '/checkout', 120, '2026-01-01T10:00:00Z'), + ('org_1', '/checkout', 900, '2026-01-01T10:01:00Z'), + ('org_1', '/search', 40, '2026-01-01T10:02:00Z'); +`) +const result = await db.query>(compiled.sql, [...compiled.parameters]) +export const rows = await Effect.runPromise(compiled.decodeRows(result.rows)) +// [{ route: "/checkout", count: 2, slow: 1, p50: 510 }, { route: "/search", count: 1, slow: 0, p50: 40 }] +await db.close() +``` + +`PGlite` stands in for any driver that takes `(sql, values)`: node-postgres, postgres.js, or +`@effect/sql-pg`'s `unsafe`. The builder never runs the query. + +## What the dialect changes + +| | ClickHouse (`clickhouseDialect`) | Postgres (`postgresDialect`) | +| --- | --- | --- | +| Identifiers | Bare: `events.OrgId` | Quoted: `"events"."OrgId"` | +| String literals | Backslash escapes: `'it\'s'` | Doubled quotes: `'it''s'` | +| Params | Written in as literals; `parameters` empty | Bound as `$1`, `$2`, … in `parameters` | +| `param.bool` | `1` / `0` | `true` / `false` | +| `param.dateTime` | `'2026-01-01 00:00:00'` (UTC, zoneless) | `'2026-01-01T00:00:00.000Z'` | +| `GROUP BY` keys | Select aliases | Select-list positions (`GROUP BY 1`) | +| Wrapped union | `SELECT * FROM (…)` | `SELECT * FROM (…) AS "__union"` | +| `.format()` | `FORMAT JSON` | Refused at compile time | + +Postgres reads a bare name in `GROUP BY` as an input column before a select alias, so +`select({ Service: lower($.Service) }).groupBy("Service")` would group by the raw column. Writing +the position instead keeps ClickHouse's meaning. + +Every string that reaches the SQL as a literal is escaped for Postgres. A value that spells the +param marker `__PARAM_` is written as an `E'…'` string with the marker hex-escaped, and a +literal that still contained it would fail the compile with `InvalidLiteral`. + +## Column types + +| Constructor | Postgres type | Decodes to | Accepts on the wire | +| --- | --- | --- | --- | +| `text`, `uuid` | `text`, `uuid` | `string` | string | +| `bool` | `boolean` | `boolean` | boolean | +| `int2`, `int4`, `int8`, `float4`, `float8`, `numeric` | same | `number` | number, numeric string, `bigint` | +| `timestamptz` | `timestamptz` | `DateTime.Utc` | `Date`, or text such as `2026-01-01 00:00:00+00` | +| `jsonb(schema?)` | `jsonb` | the schema's type (`unknown` by default) | a parsed value | +| `array(type)` | `type[]` | `ReadonlyArray` | array | +| `nullable(type)` | the same type | `T \| null` | the same, or `null` | +| `custom(sql, schema, literalSchema?)` | anything | the schema's type | whatever the schema reads | + +`int8` and `numeric` decode to `number`, so values beyond 2^53 or a double's precision lose +digits. Declare `PG.custom("int8", Schema.String)` where exact digits matter. A `timestamptz` +compared against a `Date`, a `DateTime.Utc` or a string is written as an ISO-8601 instant, +which no session time zone can reinterpret; a zoneless string is read as UTC. + +## Functions + +| Function | SQL | Notes | +| --- | --- | --- | +| `count()`, `countDistinct(x)` | `count(*)`, `count(DISTINCT x)` | | +| `countIf(c)`, `sumIf(x, c)` | `count(*) FILTER (WHERE c)`, `sum(x) FILTER (WHERE c)` | ClickHouse's `-If` combinators | +| `sum`, `avg`, `min`, `max` | same | `null` over no rows, where ClickHouse returns `0` for `sum` | +| `percentileCont(f, x)` | `percentile_cont(f) WITHIN GROUP (ORDER BY x)` | Interpolated | +| `arrayAgg(x)` | `array_agg(x)` | | +| `dateTrunc(unit, ts)` | `date_trunc(unit, ts, 'UTC')` | UTC buckets; Postgres 12+ | +| `dateBin(seconds, ts)` | `date_bin(…, ts, epoch)` | Epoch-aligned buckets; Postgres 14+ | +| `now()` | `now()` | | +| `lower`, `upper`, `length` | same | | +| `coalesce(x, fallback)` | `coalesce(x, fallback)` | No longer nullable | +| `jsonText(x, key)` | `(x ->> key)` | `null` when absent | + +The shared operators (`eq`, `in_`, `like`, `ilike`, `and`, `or`, `not`, arithmetic, `lit`) work +unchanged. The ClickHouse function catalog on the root entry (`quantile`, `toStartOfInterval`, +map subscripts, …) writes ClickHouse SQL and will not run on Postgres. + +## Known differences + +- Integer division truncates in Postgres (`7 / 2` is `3`) and promotes to a float in ClickHouse. +- `rawExpr`, `untypedExpr`, `dynamicColumn` and `outerRef` take SQL as written; quote + identifiers yourself (`"OrgId"`) when targeting Postgres. +- An outer join fills missing columns with `NULL` in Postgres and with type defaults in + ClickHouse (unless `join_use_nulls` is set). Joined columns are typed nullable either way. + +See [`design/dialects.md`](../design/dialects.md) for how dialects are built and what a new one +needs to provide. diff --git a/docs/reference.md b/docs/reference.md index d767975..5243090 100644 --- a/docs/reference.md +++ b/docs/reference.md @@ -36,6 +36,7 @@ The root barrel is curated. These are exported by the package but not from it: | `Suite`, `RunOutput`, `BenchmarkError` and benchmark contracts | `/benchmark` | | `makeHttpClient`, `makeHttpTransport`, `httpConfigFromEnv` | `/benchmark/http` | | `runCli` | `/benchmark/cli` | +| `postgresDialect`, Postgres column types and functions, Postgres-default `compile` | `/postgres` | Every column-type constructor and every expression helper is on the root as well as on its subpath. See [Running a query](./running-queries.md) for what the `/sql` statement helpers are @@ -90,8 +91,10 @@ Note `/sql` exports a `compile` (fragment → string) distinct from the root `co | `rawCompiledQuery` | `({ sql, tenantScope, reason, justification, rowSchema?, route? }) => CompiledQuery` | | `clickhouseDialect` | The default `Dialect`: params written into the SQL as ClickHouse literals | -`Dialect` and `ParamStyle` describe how params reach the server; pass one as -`options.dialect`. See [Params and compilation](./params-and-compilation.md#dialects). +`Dialect`, `DialectClauses` and `ParamStyle` describe a database: how identifiers and literals +are written, how params reach the server, and which clauses exist. Pass one as +`options.dialect`. See [Params and compilation](./params-and-compilation.md#dialects) and +[Postgres](./postgres.md). ### Params @@ -273,7 +276,7 @@ Types: `WindowSpec`, `CompiledWindowSpec`, `WindowFrameBound`, `WindowRowsFrame` **Everything else** — `Table`, `TableOptions`, `Expr`, `ColumnRef`, `Condition`, `Comparable` (what a value of a type may be compared against), `MapValueOf`, `Subquery`, `ParamMarker`, `ParamKind`, `CHQuery`, `CHUnionQuery`, `ColumnAccessor`, `JoinedColumnAccessor`, -`JoinOnCallback`, `CompiledQuery`, `CompiledQueryInput`, `CompiledQueryRowSchema`, `RowSchemaMismatch`, `TenantScope`, `Dialect`, `ParamStyle`, `FnResult`, +`JoinOnCallback`, `CompiledQuery`, `CompiledQueryInput`, `CompiledQueryRowSchema`, `RowSchemaMismatch`, `TenantScope`, `Dialect`, `DialectClauses`, `ParamStyle`, `FnResult`, `WindowFunnelMode`, `WindowSpec`, `WindowRowsFrame`, `WindowFrameBound`, `WindowOrderDirection`, `CompiledWindowSpec`. diff --git a/package.json b/package.json index 828356d..286c2f6 100644 --- a/package.json +++ b/package.json @@ -46,6 +46,10 @@ "types": "./dist/sql.d.mts", "import": "./dist/sql.mjs" }, + "./postgres": { + "types": "./dist/postgres.d.mts", + "import": "./dist/postgres.mjs" + }, "./benchmark": { "types": "./dist/benchmark/index.d.mts", "import": "./dist/benchmark/index.mjs" @@ -77,6 +81,7 @@ "@effect/platform-bun": "4.0.0", "@effect/sql-clickhouse": "4.0.0", "@effect/vitest": "4.0.0", + "@electric-sql/pglite": "0.3.15", "@types/node": "^22.10.2", "effect": "4.0.0", "expect-type": "^1.3.0", diff --git a/scripts/check-doc-examples.mjs b/scripts/check-doc-examples.mjs index cb872e1..1a4fe03 100644 --- a/scripts/check-doc-examples.mjs +++ b/scripts/check-doc-examples.mjs @@ -101,6 +101,13 @@ const escapedSql = await import("./escaped-sql") const fragments = await import("@maple-dev/effect-clickhouse/sql") assert.equal(escapedSql.predicate, "Name = " + fragments.compile(fragments.str("O'Reilly"))) assert.doesNotMatch(escapedSql.predicate, /\\[object Object\\]/) +const postgres = await import("./postgres-quickstart") +assert.deepEqual(postgres.compiled.parameters, ["org_1", "2026-01-01T00:00:00.000Z"]) +assert.ok(sql(postgres.compiled).includes('"requests"."OrgId" = $1 AND "requests"."At" >= $2')) +assert.deepEqual(postgres.rows, [ + { route: "/checkout", count: 2, slow: 1, p50: 510 }, + { route: "/search", count: 1, slow: 0, p50: 40 }, +]) console.log("Markdown example behavior checks passed") `, ) diff --git a/src/ch/compile.ts b/src/ch/compile.ts index 9993619..01e4f7d 100644 --- a/src/ch/compile.ts +++ b/src/ch/compile.ts @@ -68,6 +68,17 @@ const orderByClause = (specs: ReadonlyArray<[string, "asc" | "desc"]>): Array): string => { + const position = selected.indexOf(key) + return currentDialect().clauses.groupByAlias || position === -1 ? quoteIdent(key) : String(position + 1) +} + /** `.format()`'s value, refused for a dialect that has no `FORMAT` clause: * dropping it would hand the caller rows in a shape they did not ask for. A * defect, because the format is written in the query definition. */ @@ -747,7 +758,7 @@ function compileInner< from: fromFragment, joins, where: whereFragments, - groupBy: state.groupByKeys.map((k) => raw(quoteIdent(k))), + groupBy: state.groupByKeys.map((k) => raw(groupByKey(k, options?.selectKeys ?? keys))), // Deliberately excluded from tenant evidence: by HAVING time the // rows are already aggregated, so the scan that produced them crossed // tenants no matter what this filters out. @@ -1166,7 +1177,7 @@ function renderParams( missing.push(name) return placeholder } - const value = encodeParam(kind as ParamKind, name, params[name]) + const value = encodeParam(dialect, kind as ParamKind, name, params[name]) if (style._tag === "inline") return checkedLiteral(dialect, value, paramContext(kind, name)) const key = `${kind}\0${name}` @@ -1219,8 +1230,8 @@ const paramContext = (kind: string, name: string): string => `param '${name}' ($ * the two directions cannot drift: a `DateTime` param and a `DateTime` column * agree on the literal by construction, not by two functions being kept in sync. */ -function encodeParam(kind: ParamKind, name: string, value: unknown): unknown { - const schema = paramSchema(kind) +function encodeParam(dialect: Dialect, kind: ParamKind, name: string, value: unknown): unknown { + const schema = dialect.paramCodecs?.[kind] ?? paramSchema(kind) if (schema === undefined) { // Only reachable from a hand-written placeholder naming a kind nothing // declared: `param.of` registers its type before it can reach any SQL — diff --git a/src/ch/dialect.test.ts b/src/ch/dialect.test.ts index 79ab8e4..73ef480 100644 --- a/src/ch/dialect.test.ts +++ b/src/ch/dialect.test.ts @@ -189,7 +189,7 @@ describe("dialect identifiers and clauses", () => { ...CH.clickhouseDialect, name: "quoted", quoteIdent: (name) => `"${name.replace(/"/g, '""')}"`, - clauses: { format: false, derivedTableAlias: true }, + clauses: { format: false, derivedTableAlias: true, groupByAlias: true }, } const services = CH.table("db.services", { OrgId: CH.string, Service: CH.string }, { tenantColumn: "OrgId" }) diff --git a/src/ch/dialect.ts b/src/ch/dialect.ts index 37c314a..7bc190c 100644 --- a/src/ch/dialect.ts +++ b/src/ch/dialect.ts @@ -46,6 +46,9 @@ export interface DialectClauses { readonly format: boolean /** Whether a subquery in FROM needs an alias (`SELECT * FROM (…) AS u`). */ readonly derivedTableAlias: boolean + /** Whether GROUP BY resolves a select alias before an input column of the + * same name. Where it does not, keys are written by select-list position. */ + readonly groupByAlias: boolean } /** @@ -78,7 +81,7 @@ export const clickhouseDialect: Dialect = { literal: sqlLiteral, dateTimeLiteral: (value) => quoteClickHouseString(chDateTimeLiteral(value)), params: { _tag: "inline" }, - clauses: { format: true, derivedTableAlias: false }, + clauses: { format: true, derivedTableAlias: false, groupByAlias: true }, } // The dialect of the enclosing compile, beside the syntax installed for the diff --git a/src/ch/index.ts b/src/ch/index.ts index c66e169..7875d01 100644 --- a/src/ch/index.ts +++ b/src/ch/index.ts @@ -258,7 +258,7 @@ export { } from "./compile" // Dialects: how a compiled query's params reach the server. -export { clickhouseDialect, type Dialect, type ParamStyle } from "./dialect" +export { clickhouseDialect, type Dialect, type DialectClauses, type ParamStyle } from "./dialect" // Failures vs defects — the rule the two classes encode is on `QueryBuilderError`. export { QueryBuilderError, QueryBuilderDefect } from "./errors" diff --git a/src/pg/dialect.ts b/src/pg/dialect.ts new file mode 100644 index 0000000..67217ef --- /dev/null +++ b/src/pg/dialect.ts @@ -0,0 +1,86 @@ +// The Postgres dialect. + +import { DateTime, Schema } from "effect" +import type { Dialect } from "../ch/dialect" +import { QueryBuilderError } from "../ch/errors" +import { PARAM_MARKER_PREFIX } from "../ch/param" +import { PgTimestampLiteral, timestampLiteral } from "./types" + +const quoteIdent = (name: string): string => `"${name.replace(/"/g, '""')}"` + +/** + * A string literal. Plain `'...'` with quotes doubled, which reads backslashes + * as text (`standard_conforming_strings`, the default since Postgres 9.1). A + * value that contains the param marker is written as an `E'...'` string instead, + * the one form where the marker can be hex-escaped (`\x5F` is `_`). + */ +const quoteString = (value: string): string => { + if (value.includes("\0")) { + throw new QueryBuilderError({ + code: "InvalidLiteral", + message: "a string literal: Postgres text cannot contain a NUL character", + }) + } + if (!value.includes(PARAM_MARKER_PREFIX)) return `'${value.replace(/'/g, "''")}'` + const escaped = value + .replace(/\\/g, "\\\\") + .replace(/'/g, "''") + .replace(/__PARAM_/g, "\\x5F_PARAM_") + return `E'${escaped}'` +} + +const isPlainObject = (value: unknown): value is Record => + typeof value === "object" && + value !== null && + (Object.getPrototypeOf(value) === Object.prototype || Object.getPrototypeOf(value) === null) + +/** An encoded wire value as a Postgres literal. */ +const literal = (value: unknown, context: string): string => { + if (value === null) return "NULL" + if (typeof value === "string") return quoteString(value) + if (typeof value === "boolean") return value ? "TRUE" : "FALSE" + if (typeof value === "bigint") return String(value) + if (typeof value === "number") { + if (!Number.isFinite(value)) { + throw new QueryBuilderError({ code: "InvalidLiteral", message: `${context}: ${value} has no Postgres literal` }) + } + return String(value) + } + // `ARRAY[]` needs a cast to say its element type; the untyped `'{}'` takes + // the type of whatever it is compared against. + if (Array.isArray(value)) { + return value.length === 0 ? "'{}'" : `ARRAY[${value.map((element) => literal(element, context)).join(", ")}]` + } + // A record reaches here from a jsonb codec that did not stringify it. + if (isPlainObject(value)) return quoteString(JSON.stringify(value)) + throw new QueryBuilderError({ + code: "InvalidLiteral", + message: `${context}: cannot write ${typeof value} as a Postgres literal`, + }) +} + +/** `param.dateTimeSeconds`: the same instant, floored to whole seconds. */ +const timestampSeconds = timestampLiteral((epochMillis) => new Date(Math.floor(epochMillis / 1000) * 1000).toISOString()) + +/** + * Postgres: double-quoted identifiers, standard string literals, and params + * bound to `$1`, `$2`, … and returned in `CompiledQuery.parameters`. + * + * Portable param kinds encode for Postgres: `param.bool` binds a boolean + * rather than ClickHouse's `1`/`0`, and `param.dateTime` an ISO-8601 instant + * rather than a zoneless string a session time zone could reinterpret. + */ +export const postgresDialect: Dialect = { + name: "postgres", + quoteIdent, + quoteString, + literal, + dateTimeLiteral: (value) => `TIMESTAMPTZ ${quoteString(DateTime.formatIso(value))}`, + params: { _tag: "bind", placeholder: (index) => `$${index}`, reuse: true }, + clauses: { format: false, derivedTableAlias: true, groupByAlias: false }, + paramCodecs: { + bool: Schema.Boolean, + dateTime: PgTimestampLiteral, + dateTimeSeconds: timestampSeconds, + }, +} diff --git a/src/pg/functions.ts b/src/pg/functions.ts new file mode 100644 index 0000000..4b13813 --- /dev/null +++ b/src/pg/functions.ts @@ -0,0 +1,120 @@ +// Postgres functions. +// +// The shared operators (`eq`, `and`, `in_`, `like`, arithmetic, `not`) come +// from the builder and render the same everywhere. These are the ones whose +// Postgres spelling or result type differs from ClickHouse's: `count(*)` rather +// than `count()`, `FILTER (WHERE …)` rather than `-If` combinators, and +// aggregates over no rows returning NULL rather than 0. + +import { Schema, type DateTime } from "effect" +import { QueryBuilderDefect } from "../ch/errors" +import { makeExpr, type Condition, type Expr } from "../ch/expr" +import { schemaOf, withoutNull } from "../ch/define-fn" +import { compile, lazy, raw, str } from "../sql/sql-fragment" +import * as T from "./types" + +const sql = (expr: Expr | Condition): string => compile(expr.toFragment()) + +const nullableNumber = Schema.NullOr(T.PgNumber) as Schema.Codec +const int8 = T.int8.schema as Schema.Codec + +// Aggregates + +/** `count(*)`. */ +export const count = (): Expr => makeExpr(raw("count(*)"), int8) + +/** `count(DISTINCT expr)`. */ +export const countDistinct = (expr: Expr): Expr => + makeExpr(lazy(() => `count(DISTINCT ${sql(expr)})`), int8) + +/** `count(*) FILTER (WHERE condition)`: ClickHouse's `countIf`. */ +export const countIf = (condition: Condition): Expr => + makeExpr(lazy(() => `count(*) FILTER (WHERE ${sql(condition)})`), int8) + +/** `sum(expr)`. NULL over no rows, and a string for int8/numeric inputs on the + * wire, which the result codec reads as a number. */ +export const sum = (expr: Expr): Expr => + makeExpr(lazy(() => `sum(${sql(expr)})`), nullableNumber) + +/** `sum(expr) FILTER (WHERE condition)`: ClickHouse's `sumIf`. */ +export const sumIf = (expr: Expr, condition: Condition): Expr => + makeExpr(lazy(() => `sum(${sql(expr)}) FILTER (WHERE ${sql(condition)})`), nullableNumber) + +/** `avg(expr)`. NULL over no rows. */ +export const avg = (expr: Expr): Expr => + makeExpr(lazy(() => `avg(${sql(expr)})`), nullableNumber) + +const nullableOf = (expr: Expr): Schema.Codec | undefined => { + const schema = schemaOf(expr) + return schema === undefined ? undefined : (Schema.NullOr(schema) as Schema.Codec) +} + +/** `min(expr)`, decoding as `expr` does. NULL over no rows. */ +export const min = (expr: Expr): Expr => makeExpr(lazy(() => `min(${sql(expr)})`), nullableOf(expr)) + +/** `max(expr)`, decoding as `expr` does. NULL over no rows. */ +export const max = (expr: Expr): Expr => makeExpr(lazy(() => `max(${sql(expr)})`), nullableOf(expr)) + +/** `percentile_cont(fraction) WITHIN GROUP (ORDER BY expr)`: an interpolated + * quantile, ClickHouse's `quantileExact` family. */ +export const percentileCont = (fraction: number, expr: Expr): Expr => { + if (!(fraction >= 0 && fraction <= 1)) { + throw new QueryBuilderDefect({ message: `percentileCont: fraction must be within [0, 1], got ${fraction}` }) + } + return makeExpr(lazy(() => `percentile_cont(${fraction}) WITHIN GROUP (ORDER BY ${sql(expr)})`), nullableNumber) +} + +/** `array_agg(expr)`. NULL over no rows. */ +export const arrayAgg = (expr: Expr): Expr | null> => { + const element = schemaOf(expr) + return makeExpr( + lazy(() => `array_agg(${sql(expr)})`), + element === undefined ? undefined : (Schema.NullOr(Schema.Array(element)) as Schema.Codec | null, unknown>), + ) +} + +// Time + +const timestamptz = T.timestamptz.schema as Schema.Codec + +export type DateTruncUnit = "second" | "minute" | "hour" | "day" | "week" | "month" | "quarter" | "year" + +/** `date_trunc(unit, ts, 'UTC')`: buckets in UTC whatever the session time + * zone, as ClickHouse's `toStartOf*` functions do. Postgres 12+. */ +export const dateTrunc = (unit: DateTruncUnit, ts: Expr): Expr => + makeExpr(lazy(() => `date_trunc(${compile(str(unit))}, ${sql(ts)}, 'UTC')`), timestamptz) + +/** `date_bin(seconds, ts, epoch)`: fixed-width buckets aligned to the Unix + * epoch, ClickHouse's `toStartOfInterval`. Postgres 14+. */ +export const dateBin = (seconds: number, ts: Expr): Expr => { + if (!(Number.isSafeInteger(seconds) && seconds > 0)) { + throw new QueryBuilderDefect({ message: `dateBin: bucket width must be a positive whole number of seconds, got ${seconds}` }) + } + return makeExpr( + lazy(() => `date_bin(make_interval(secs => ${seconds}), ${sql(ts)}, TIMESTAMPTZ '1970-01-01 00:00:00+00')`), + timestamptz, + ) +} + +/** `now()`: the transaction's start time. */ +export const now = (): Expr => makeExpr(raw("now()"), timestamptz) + +// Strings and values + +const text = T.text.schema as Schema.Codec + +export const lower = (expr: Expr): Expr => makeExpr(lazy(() => `lower(${sql(expr)})`), text) +export const upper = (expr: Expr): Expr => makeExpr(lazy(() => `upper(${sql(expr)})`), text) +export const length = (expr: Expr): Expr => + makeExpr(lazy(() => `length(${sql(expr)})`), T.int4.schema as Schema.Codec) + +/** `coalesce(expr, fallback)`, no longer nullable. */ +export const coalesce = (expr: Expr, fallback: Expr): Expr => + makeExpr( + lazy(() => `coalesce(${sql(expr)}, ${sql(fallback)})`), + schemaOf(fallback) ?? withoutNull(schemaOf(expr)), + ) + +/** `expr ->> key`: a jsonb field as text, NULL when it is absent. */ +export const jsonText = (expr: Expr, key: string): Expr => + makeExpr(lazy(() => `(${sql(expr)} ->> ${compile(str(key))})`), Schema.NullOr(Schema.String) as Schema.Codec) diff --git a/src/pg/postgres.test.ts b/src/pg/postgres.test.ts new file mode 100644 index 0000000..1dd604d --- /dev/null +++ b/src/pg/postgres.test.ts @@ -0,0 +1,216 @@ +// The Postgres dialect against a real Postgres: PGlite (Postgres 17 compiled to +// WASM), in-process. Every query here is compiled by the builder and executed, +// so a test passes only if Postgres parses the SQL, binds the params, and the +// rows decode through the declared types. + +import { PGlite } from "@electric-sql/pglite" +import { afterAll, beforeAll, describe, expect, it } from "@effect/vitest" +import { DateTime, Effect } from "effect" +import type { CompiledQuery } from "../ch/compile" +import * as CH from "../index" +import * as PG from "../postgres" + +const db = new PGlite() + +const events = CH.table( + "events", + { + OrgId: PG.text, + Service: PG.text, + Count: PG.int8, + Timestamp: PG.timestamptz, + Ok: PG.bool, + Attrs: PG.jsonb(), + }, + { tenantColumn: "OrgId" }, +) + +const services = CH.table("app.services", { OrgId: PG.text, Service: PG.text, Team: PG.nullable(PG.text) }, { tenantColumn: "OrgId" }) + +const tricky = "it's \\ a __PARAM_string_orgId__ value" + +beforeAll(async () => { + await db.exec(` + CREATE SCHEMA app; + CREATE TABLE events ("OrgId" text, "Service" text, "Count" int8, "Timestamp" timestamptz, "Ok" boolean, "Attrs" jsonb); + CREATE TABLE app.services ("OrgId" text, "Service" text, "Team" text); + INSERT INTO events VALUES + ('org_1', 'api', 10, '2026-01-01T00:00:10Z', true, '{"region": "eu"}'), + ('org_1', 'api', 20, '2026-01-01T00:04:59Z', false, '{"region": "eu"}'), + ('org_1', 'web', 5, '2026-01-01T00:05:00Z', true, '{"region": "us"}'), + ('org_1', $$${tricky}$$, 1, '2026-01-01T01:00:00Z', true, '{}'), + ('org_2', 'api', 99, '2026-01-01T00:00:00Z', true, '{"region": "eu"}'); + INSERT INTO app.services VALUES ('org_1', 'api', 'core'), ('org_1', 'web', NULL); + `) +}) + +afterAll(async () => { + await db.close() +}) + +const run = async (compiled: CompiledQuery): Promise> => { + const result = await db.query>(compiled.sql, [...compiled.parameters]) + return Effect.runPromise(compiled.decodeRows(result.rows)) +} + +describe("postgres dialect", () => { + it("aggregates per group with bound params, FILTER, and quoted identifiers", async () => { + const query = CH.from(events) + .select(($) => ({ + service: $.Service, + events: PG.count(), + failures: PG.countIf($.Ok.eq(false)), + total: PG.sum($.Count), + average: PG.avg($.Count), + })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId")), $.Service.in_("api", "web")]) + .groupBy("service") + .orderBy(["service", "asc"]) + const compiled = PG.compileUnsafe(query, { orgId: "org_1" }) + + expect(compiled.sql).toContain(`"OrgId" = $1`) + expect(compiled.parameters).toEqual(["org_1"]) + expect(compiled.tenantScope).toBe("single-tenant") + expect(await run(compiled)).toEqual([ + { service: "api", events: 2, failures: 1, total: 30, average: 15 }, + { service: "web", events: 1, failures: 0, total: 5, average: 5 }, + ]) + }) + + it("decodes timestamptz and buckets with dateBin and dateTrunc in UTC", async () => { + const query = CH.from(events) + .select(($) => ({ + bucket: PG.dateBin(300, $.Timestamp), + hour: PG.dateTrunc("hour", $.Timestamp), + events: PG.count(), + })) + .where(($) => [ + $.OrgId.eq("org_1"), + $.Timestamp.gte(CH.param.dateTime("start")), + $.Timestamp.lt(CH.param.dateTime("end")), + ]) + .groupBy("bucket", "hour") + .orderBy(["bucket", "asc"]) + const rows = await run( + PG.compileUnsafe(query, { + start: new Date("2026-01-01T00:00:00Z"), + end: DateTime.makeUnsafe("2026-01-01T00:30:00Z"), + }), + ) + expect(rows.map((row) => [DateTime.formatIso(row.bucket), DateTime.formatIso(row.hour), row.events])).toEqual([ + ["2026-01-01T00:00:00.000Z", "2026-01-01T00:00:00.000Z", 2], + ["2026-01-01T00:05:00.000Z", "2026-01-01T00:00:00.000Z", 1], + ]) + }) + + it("joins a schema-qualified table, reads a subquery, and nulls a missing join value", async () => { + const perService = CH.from(events) + .select(($) => ({ OrgId: $.OrgId, Service: $.Service, total: PG.sum($.Count) })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId"))]) + .groupBy("OrgId", "Service") + const query = CH.fromQuery(perService, "per") + .innerJoin(services, "s", (per, s) => per.Service.eq(s.Service).and(per.OrgId.eq(s.OrgId))) + .select(($) => ({ service: $.Service, team: PG.coalesce($.s.Team, CH.lit("none")), total: $.total })) + .orderBy(["service", "asc"]) + const compiled = PG.compileUnsafe(query, { orgId: "org_1" }) + expect(compiled.sql).toContain(`INNER JOIN "app"."services" AS "s"`) + expect(await run(compiled)).toEqual([ + { service: "api", team: "core", total: 30 }, + { service: "web", team: "none", total: 5 }, + ]) + }) + + it("runs a CTE and an ordered union, which Postgres needs aliased", async () => { + const recent = CH.table("recent", { OrgId: PG.text, Count: PG.int8 }) + const cte = CH.from(recent) + .withCTE( + "recent", + CH.from(events) + .select(($) => ({ OrgId: $.OrgId, Count: $.Count })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId"))]), + ) + .select(($) => ({ total: PG.sum($.Count) })) + expect(await run(PG.compileUnsafe(cte, { orgId: "org_1" }))).toEqual([{ total: 36 }]) + + const branch = (org: string) => + CH.from(events) + .select(($) => ({ org: $.OrgId, total: PG.sum($.Count) })) + .where(($) => [$.OrgId.eq(CH.param.string(org))]) + .groupBy("org") + const union = PG.compileUnionUnsafe( + CH.unionAll(branch("first"), branch("second")).orderBy(["total", "desc"]), + { first: "org_1", second: "org_2" }, + ) + expect(union.parameters).toEqual(["org_1", "org_2"]) + expect(await run(union)).toEqual([ + { org: "org_2", total: 99 }, + { org: "org_1", total: 36 }, + ]) + }) + + // The value spells a placeholder and carries a quote and a backslash, and is + // matched three ways: as a bound param, as a column literal, and through + // LIKE. Each has to reach Postgres as data, not as SQL or as a placeholder. + it("keeps quotes, backslashes and the param marker inside values", async () => { + const byParam = CH.from(events) + .select(($) => ({ count: $.Count })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId")), $.Service.eq(CH.param.string("service"))]) + expect(await run(PG.compileUnsafe(byParam, { orgId: "org_1", service: tricky }))).toEqual([{ count: 1 }]) + + const byLiteral = CH.from(events) + .select(($) => ({ count: $.Count })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId")), $.Service.eq(tricky), $.Service.like("it's \\\\%")]) + const compiled = PG.compileUnsafe(byLiteral, { orgId: "org_1" }) + expect(compiled.sql).toContain(`E'it''s \\\\ a \\x5F_PARAM_string_orgId__ value'`) + expect(compiled.parameters).toEqual(["org_1"]) + expect(await run(compiled)).toEqual([{ count: 1 }]) + }) + + it("binds booleans and compares jsonb", async () => { + const query = CH.from(events) + .select(($) => ({ region: PG.jsonText($.Attrs, "region"), regions: PG.arrayAgg($.Service) })) + .where(($) => [ + $.OrgId.eq("org_1"), + $.Ok.eq(CH.param.bool("ok")), + $.Attrs.neq({ region: "us" }), + ]) + .groupBy("region") + .orderBy(["region", "asc"]) + const compiled = PG.compileUnsafe(query, { ok: true }) + expect(compiled.parameters).toEqual([true]) + expect(await run(compiled)).toEqual([ + { region: "eu", regions: ["api"] }, + { region: null, regions: [tricky] }, + ]) + }) + + it("computes an interpolated percentile", async () => { + const query = CH.from(events) + .select(($) => ({ p50: PG.percentileCont(0.5, $.Count), lowest: PG.min($.Count) })) + .where(($) => [$.OrgId.eq("org_1")]) + expect(await run(PG.compileUnsafe(query, {}))).toEqual([{ p50: 7.5, lowest: 1 }]) + }) + + it("refuses FORMAT and fails a missing param before reaching Postgres", () => { + const query = CH.from(events) + .select(($) => ({ count: $.Count })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId"))]) + expect(() => PG.compileUnsafe(query.format("JSON"), { orgId: "o" })).toThrow(/no FORMAT clause/) + expect(() => PG.compileUnsafe(query, {})).toThrow(/no value given for param 'orgId'/) + }) +}) + +describe("postgres group by", () => { + // An alias that shadows an input column: Postgres would group a bare + // `GROUP BY "Service"` by the raw column, splitting 'api' and 'API'. + it("groups by the selected expression, not a same-named input column", async () => { + await db.exec(`INSERT INTO events VALUES ('org_3', 'API', 1, now(), true, '{}'), ('org_3', 'api', 2, now(), true, '{}')`) + const query = CH.from(events) + .select(($) => ({ Service: PG.lower($.Service), total: PG.sum($.Count) })) + .where(($) => [$.OrgId.eq("org_3")]) + .groupBy("Service") + const compiled = PG.compileUnsafe(query, {}) + expect(compiled.sql).toContain("GROUP BY 1") + expect(await run(compiled)).toEqual([{ Service: "api", total: 3 }]) + }) +}) diff --git a/src/pg/types.ts b/src/pg/types.ts new file mode 100644 index 0000000..f5a6cca --- /dev/null +++ b/src/pg/types.ts @@ -0,0 +1,156 @@ +// Postgres Column Types +// +// The same descriptor as the ClickHouse types (`CHType`): a SQL type name, the +// codec its wire value decodes with, and the codec a compared literal encodes +// with. Only the codecs differ. Postgres drivers disagree on wire forms (int8 and +// numeric arrive as strings from node-postgres and postgres.js, timestamptz as a +// `Date` or a string depending on parser settings), so each codec accepts every +// form a common driver sends, the way `CHNumber` accepts a quoted 64-bit integer. + +import { DateTime, Schema, SchemaGetter } from "effect" +import { chDateTimeToIso, custom, type CHType, type InferEncoded, type InferTS } from "../ch/types" + +/** A Postgres column type. The ClickHouse descriptor under a dialect-neutral name. */ +export type PgType = CHType + +/** A number as any Postgres driver sends one: a number, a numeric string, or a + * `bigint`. Decodes to `number`, so an int8 beyond 2^53 loses precision; declare + * `custom("int8", Schema.String)` where that matters. */ +export const PgNumber: Schema.Codec = Schema.Union([ + Schema.Finite, + Schema.FiniteFromString, + Schema.BigInt.pipe( + Schema.decodeTo(Schema.Finite, { + decode: SchemaGetter.transform((value: bigint) => Number(value)), + encode: SchemaGetter.transform((value: number) => globalThis.BigInt(value)), + }), + ), +]) + +/** + * Postgres's text form of a timestamptz (`2026-01-01 00:00:00.25+00`) as + * ISO-8601, which `Date` parses the same way on every runtime. A zoneless value + * is read as UTC, the same rule the ClickHouse types apply. + */ +export const pgTimestampToIso = (value: string): string => { + const trimmed = value.trim() + const zoned = /^(\d{4}-\d{2}-\d{2})[ T](\d{2}:\d{2}:\d{2}(?:\.\d+)?)([+-]\d{2})(?::?(\d{2}))?$/.exec(trimmed) + if (zoned) return `${zoned[1]}T${zoned[2]}${zoned[3]}:${zoned[4] ?? "00"}` + return chDateTimeToIso(trimmed) +} + +const isTimestamp = Schema.makeFilter( + (value: string) => + Number.isNaN(Date.parse(pgTimestampToIso(value))) + ? `\`${value}\` is not a timestamp: expected ISO-8601 or \`YYYY-MM-DD hh:mm:ss[+zz]\`` + : undefined, + { title: "postgresTimestamp" }, +) + +/** A timestamptz from its text form. */ +const TimestampFromString: Schema.Codec = Schema.String.pipe( + Schema.check(isTimestamp), + Schema.decodeTo(Schema.DateTimeUtc, { + decode: SchemaGetter.transform((value: string) => DateTime.makeUnsafe(pgTimestampToIso(value))), + encode: SchemaGetter.transform((value: DateTime.Utc) => DateTime.formatIso(value)), + }), +) + +/** A timestamptz as a driver sends it: a `Date` (the usual parse) or text. */ +const PgTimestamptz: Schema.Codec = Schema.Union([ + Schema.DateTimeUtcFromDate, + TimestampFromString, +]) + +/** + * Everything a timestamptz can be compared against (a `DateTime.Utc`, a `Date`, + * or a timestamp string), written as an ISO-8601 instant by `format`, which + * receives epoch milliseconds. Unlike a zoneless literal, an instant cannot be + * read in the session's time zone. + */ +export const timestampLiteral = (format: (epochMillis: number) => string): Schema.Codec => + Schema.Union([ + Schema.String.pipe( + Schema.decodeTo(Schema.DateTimeUtc, { + decode: SchemaGetter.transform((value: string) => DateTime.makeUnsafe(pgTimestampToIso(value))), + encode: SchemaGetter.transform((value: DateTime.Utc) => format(DateTime.toEpochMillis(value))), + }), + ), + Schema.String.pipe( + Schema.decodeTo(Schema.Date, { + decode: SchemaGetter.transform((value: string) => new Date(value)), + encode: SchemaGetter.transform((value: Date) => format(value.getTime())), + }), + ), + Schema.String.pipe( + Schema.check(isTimestamp), + Schema.decodeTo(Schema.String, { + decode: SchemaGetter.transform((value: string) => value), + encode: SchemaGetter.transform((value: string) => format(Date.parse(pgTimestampToIso(value)))), + }), + ), + ]) as Schema.Codec + +const isoInstant = (epochMillis: number): string => new Date(epochMillis).toISOString() + +export const PgTimestampLiteral: Schema.Codec = timestampLiteral(isoInstant) + +// Scalars + +export const text: PgType<"text", string> = custom("text", Schema.String) +export const uuid: PgType<"uuid", string> = custom("uuid", Schema.String) +export const bool: PgType<"boolean", boolean> = custom("boolean", Schema.Boolean) +export const int2: PgType<"int2", number, number | string | bigint> = custom("int2", PgNumber) +export const int4: PgType<"int4", number, number | string | bigint> = custom("int4", PgNumber) +export const int8: PgType<"int8", number, number | string | bigint> = custom("int8", PgNumber) +export const float4: PgType<"float4", number, number | string | bigint> = custom("float4", PgNumber) +export const float8: PgType<"float8", number, number | string | bigint> = custom("float8", PgNumber) +/** Decodes to `number`; precision beyond a double is lost. */ +export const numeric: PgType<"numeric", number, number | string | bigint> = custom("numeric", PgNumber) +export const timestamptz: PgType<"timestamptz", DateTime.Utc, Date | string> = custom( + "timestamptz", + PgTimestamptz, + PgTimestampLiteral, +) + +/** + * A jsonb column, decoded with `schema` (anything, by default). Drivers parse + * jsonb themselves, so rows arrive as values; a compared literal is written as + * JSON text, which Postgres reads back as jsonb. + */ +export const jsonb = ( + schema: Schema.Codec = Schema.Unknown as Schema.Codec, +): PgType<"jsonb", A, unknown> => + custom("jsonb", schema as Schema.Codec, Schema.fromJsonString(schema) as Schema.Codec) + +// Compound types + +export type PgArray> = PgType< + "Array", + ReadonlyArray>, + ReadonlyArray> +> + +export type PgNullable> = PgType< + "Nullable", + InferTS | null, + InferEncoded | null +> + +export const array = >(e: E): PgArray => ({ + _tag: "Array", + sql: `${e.sql}[]`, + schema: Schema.Array(e.schema), + literalSchema: Schema.Array(e.literalSchema) as Schema.Codec, + element: e, +}) as PgArray + +export const nullable = >(t: T): PgNullable => ({ + _tag: "Nullable", + sql: t.sql, + schema: Schema.NullOr(t.schema), + literalSchema: Schema.NullOr(t.literalSchema) as Schema.Codec, + element: t, +}) as PgNullable + +export { custom } diff --git a/src/postgres.ts b/src/postgres.ts new file mode 100644 index 0000000..66c151f --- /dev/null +++ b/src/postgres.ts @@ -0,0 +1,34 @@ +// @maple-dev/effect-clickhouse/postgres +// +// Postgres for the same query builder. Build queries with the root entry +// (`table`, `from`, `param`, `unionAll`, the shared operators); declare +// columns with the types here and compile with the `compile` here, which +// defaults to the Postgres dialect. See docs/postgres.md. + +import { + compileCH, + compileCHUnsafe, + compileUnion as compileUnionCH, + compileUnionUnsafe as compileUnionUnsafeCH, +} from "./ch/compile" +import { postgresDialect } from "./pg/dialect" + +export { postgresDialect } +export * from "./pg/types" +export * from "./pg/functions" + +/** `compile` from the root entry, for Postgres unless `options.dialect` says otherwise. */ +export const compile: typeof compileCH = (query, params, options) => + compileCH(query, params, { ...options, dialect: options?.dialect ?? postgresDialect }) + +/** `compileUnsafe` from the root entry, for Postgres. */ +export const compileUnsafe: typeof compileCHUnsafe = (query, params, options) => + compileCHUnsafe(query, params, { ...options, dialect: options?.dialect ?? postgresDialect }) + +/** `compileUnion` from the root entry, for Postgres. */ +export const compileUnion: typeof compileUnionCH = (union, params, options) => + compileUnionCH(union, params, { ...options, dialect: options?.dialect ?? postgresDialect }) + +/** `compileUnionUnsafe` from the root entry, for Postgres. */ +export const compileUnionUnsafe: typeof compileUnionUnsafeCH = (union, params, options) => + compileUnionUnsafeCH(union, params, { ...options, dialect: options?.dialect ?? postgresDialect }) diff --git a/tsdown.config.ts b/tsdown.config.ts index f2f473c..edee4c9 100644 --- a/tsdown.config.ts +++ b/tsdown.config.ts @@ -6,6 +6,7 @@ export default defineConfig({ expr: "./src/expr.ts", types: "./src/types.ts", sql: "./src/sql/index.ts", + postgres: "./src/postgres.ts", "benchmark/index": "./src/benchmark/index.ts", "benchmark/http": "./src/benchmark/http.ts", "benchmark/cli": "./src/benchmark/cli.ts",