diff --git a/CHANGELOG.md b/CHANGELOG.md index 878d3c7..a54a655 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,18 @@ # Changelog +## Unreleased + +- Add `Dialect`: how a compiled query writes identifiers and literals, binds params, and + which clauses exist. `compile` and friends take `options.dialect`; `clickhouseDialect` is the + default and its output is unchanged. +- Add `CompiledQuery.parameters`: the values a binding dialect sends beside `sql`, empty for + ClickHouse. Code that builds a `CompiledQuery` by hand must now supply it. +- Add the `./postgres` entry point: `postgresDialect` (quoted identifiers, `$n` binding), + Postgres column types and functions, and a `compile` that defaults to Postgres. +- A literal that would contain the param marker `__PARAM_` now fails the compile with + `InvalidLiteral` instead of relying on each dialect's escaping. +- `compileUnionUnsafe` no longer accepts the internal `enclosingCtes` option. + ## 0.2.0 - Require Effect `^4.0.0`. Effect 4.0.0 moved `effect/unstable/*` to `effect/*` diff --git a/README.md b/README.md index 5d604c8..48d7234 100644 --- a/README.md +++ b/README.md @@ -151,6 +151,7 @@ Full guides live in [`docs/`](./docs/README.md): | [Decoding results](./docs/decoding-results.md) | `rowSchema`, `decodeRows`, decode errors | | [Running a query](./docs/running-queries.md) | Executing the SQL with a real client, wire settings, `SETTINGS` | | [Tenant scoping](./docs/tenant-scoping.md) | `tenantColumn`, what marks a query scoped, `crossTenant()` | +| [Postgres](./docs/postgres.md) | The Postgres dialect, its column types and functions | | [Extending the DSL](./docs/extending.md) | `defineFn`, raw escape hatches, handwritten SQL | | [API reference](./docs/reference.md) | Full export catalog by module, plus error types | @@ -166,6 +167,7 @@ regressions live in [`src/docs-examples.test.ts`](./src/docs-examples.test.ts). | `@maple-dev/effect-clickhouse/types` | Column-type constructors (`string`, `uint64`, `dateTime`, `map`, `array`, `nullable`, …) and the `CH*` type descriptors. | | `@maple-dev/effect-clickhouse/expr` | Kitchen-sink namespace: every expression helper plus all ClickHouse functions under their raw names (`min_`, `toString_`, `toStartOfInterval`, `dynamicColumn`, …). Handy for `import * as CH`. | | `@maple-dev/effect-clickhouse/sql` | The low-level `SqlFragment` AST (`raw`, `ident`, `compile`, …) for hand-rolling fragments. | +| `@maple-dev/effect-clickhouse/postgres` | The Postgres dialect: `postgresDialect`, Postgres column types and functions, and a `compile` that defaults to Postgres. | ## Extending with custom functions diff --git a/bun.lock b/bun.lock index e42b1b9..8cc4b6c 100644 --- a/bun.lock +++ b/bun.lock @@ -10,6 +10,7 @@ "@effect/platform-bun": "4.0.0", "@effect/sql-clickhouse": "4.0.0", "@effect/vitest": "4.0.0", + "@electric-sql/pglite": "0.3.15", "@types/node": "^22.10.2", "effect": "4.0.0", "expect-type": "^1.3.0", @@ -42,6 +43,8 @@ "@effect/vitest": ["@effect/vitest@4.0.0", "", { "peerDependencies": { "effect": "^4.0.0", "vitest": ">=5.0.0 <6.0.0" } }, "sha512-kGElCPFGtP+CgMCwV/HjJWqtE3zYo6I1+8MSA54l+EI24lPU9p2Efa+TBp1jfkg9j711KKzEqgSAy3VzTsb76Q=="], + "@electric-sql/pglite": ["@electric-sql/pglite@0.3.15", "", {}, "sha512-Cj++n1Mekf9ETfdc16TlDi+cDDQF0W7EcbyRHYOAeZdsAe8M/FJg18itDTSwyHfar2WIezawM9o0EKaRGVKygQ=="], + "@jridgewell/sourcemap-codec": ["@jridgewell/sourcemap-codec@1.6.0", "", {}, "sha512-T7jf+5zgsZHwNJ4lvQ7/aezbyk0nNX+zJVWpmHA7VYsEx7a7qr5Rg5IbtJFqkgze5Y2sruq1RUY8Q837Od7iFw=="], "@oxc-project/types": ["@oxc-project/types@0.148.0", "", {}, "sha512-Nm4s/jB+4FpFsPhWGEC4h7rzksesmtnMXomo6rCMcg/b8zLQuOziRgkCS1fxDCXOlJB/6Q8oABOZ/OP6RIPj9A=="], diff --git a/design/dialects.md b/design/dialects.md index f1ddc1c..055f2ee 100644 --- a/design/dialects.md +++ b/design/dialects.md @@ -1,6 +1,8 @@ # Dialects: one builder, several databases -Status: steps 1 and 2 landed (params and literals go through a `Dialect`). Steps 3 to 6 are open. +Status: done. ClickHouse and Postgres both compile from the same builder; see +[`docs/postgres.md`](../docs/postgres.md). Steps 4 and 5 landed differently from the plan, as +noted below. ## Goal @@ -15,14 +17,14 @@ Both were read from source (Kysely 0.28.17, Drizzle 1.0.0-rc.5). | Idea | Where it comes from | How it lands here | | --- | --- | --- | -| Builders produce a config object; a dialect turns it into SQL | Drizzle `PgDialect.buildSelectQuery(config)` | `CHQueryState` already is that config. `compile.ts` becomes the ClickHouse dialect's `buildSelect` | -| SQL as a chunk tree rendered at the end with `escapeName` / `escapeParam` / `escapeString` | Drizzle `SQL` chunks + `BuildQueryConfig` | `SqlFragment` gets a real `Param` chunk; `Str`/`Lazy` stop rendering ClickHouse text early | +| Builders produce a config object; a dialect turns it into SQL | Drizzle `PgDialect.buildSelectQuery(config)` | `CHQueryState` already is that config; `compile.ts` reads the dialect where clauses differ (step 5) | +| SQL as a chunk tree rendered at the end with `escapeName` / `escapeParam` / `escapeString` | Drizzle `SQL` chunks + `BuildQueryConfig` | `Str` and `Ident` render through the installed dialect; params stay text placeholders resolved once per statement | | Inline vs bound params as a switch | Drizzle `inlineParams` | `ParamStyle`: `inline` (ClickHouse today) or `bind` | | A small override surface per dialect | Kysely `DefaultQueryCompiler` hooks (MySQL overrides ~10 methods) | The `Dialect` interface grows hook by hook, never a copy of the compiler | -| Capability flags instead of dialect checks | Kysely `DialectAdapter` (`supportsReturning`, ...) | Flags such as `supportsFilterClause`, `aliasInWhere`, `limitBy` | +| Capability flags instead of dialect checks | Kysely `DialectAdapter` (`supportsReturning`, ...) | `Dialect.clauses`: `format`, `derivedTableAlias`, `groupByAlias` | | A compiled query that keeps its structure | Kysely `CompiledQuery.query` | `CompiledQuery` already carries tenant scope and row schema; `parameters` added in step 1 | -| Rewrites as passes over the tree | Kysely plugins (`transformQuery` / `transformResult`) | Tenant-scope proof and empty-`IN` handling as passes over the state | -| Logical type separate from how a driver sends it | Drizzle v1 codecs, `refineGenericPgCodecs` per driver | `CHType` splits into a type and a transport codec (step 3) | +| Rewrites as passes over the tree | Kysely plugins (`transformQuery` / `transformResult`) | Not adopted yet; tenant scope is still derived inside `compile` | +| Logical type separate from how a driver sends it | Drizzle v1 codecs, `refineGenericPgCodecs` per driver | Per-dialect type sets on the shared descriptor, codecs tolerant of every common driver (step 4); `Dialect.paramCodecs` for params | | Shared operators, per-dialect function catalogs | Drizzle `sql/expressions` vs `pg-core` / `mysql-core` | `eq`, `and`, `in_` in core; `countIf`, `percentileCont` per dialect | What we deliberately do not copy: @@ -47,23 +49,40 @@ What we deliberately do not copy: have no dialect argument. The `__PARAM_` safety is now enforced rather than assumed: a literal that contains the marker fails with `InvalidLiteral`, so a dialect with weaker escaping cannot turn a value into a placeholder. -3. **Identifier quoting.** Postgres folds unquoted names to lower case, so `OrgId` must be - written `"OrgId"`. Column refs, aliases, qualified names, `groupBy` keys and `orderBy` - specs are raw text today (`raw(name)` in `expr.ts`, `orderByClause` in `compile.ts`), so - this is its own step: column refs become `Ident` fragments that carry their qualifier. -4. **Split `CHType`.** A logical type (`sql` name, TS type) plus a codec for the transport. The - UInt64-as-string rule belongs to ClickHouse's `FORMAT JSON` over HTTP, not to ClickHouse; - the native client sends something else, and Postgres drivers send int8 as a string and - timestamptz as a `Date`. -5. **Move `compile.ts` into `ClickHouseDialect.buildSelect(state)`.** `compileQuery` and the - terminal clauses (`FORMAT`, `SETTINGS`, `LIMIT BY`) become ClickHouse-only. -6. **Postgres dialect.** `pg.T` column types, `$n` binding, `"ident"` quoting, a function - catalog covering the common cases (`FILTER (WHERE ...)`, `date_bin`, `percentile_cont`, - jsonb access), and capability flags for what ClickHouse allows and Postgres does not - (select aliases in `WHERE`/`HAVING`, default values instead of `NULL` in outer joins). +3. **Identifier quoting (done).** Column refs are `Ident` fragments carrying their qualifier + (`ident(column, "alias")`), so a ClickHouse `Nested` column such as `Events.Name` is never + split while `db.table` qualifiers are quoted per segment. Tables, aliases, CTE names, and + group/order keys go through `Dialect.quoteIdent` too. Untyped boolean and DateTime + literals moved behind the dialect in the same step (`dateTimeLiteral`). +4. **Column types (done, differently).** The plan was to split `CHType` into a logical type + plus a transport codec. Postgres turned out not to need the split: its types reuse the + `CHType` descriptor (`PgType` is an alias) with codecs that accept every wire form a common + driver sends (int8 as number, string or `bigint`; timestamptz as `Date` or text), the same + approach `CHNumber` already takes for quoted 64-bit integers. Portable param kinds that + must encode differently (`param.bool`, `param.dateTime`) are re-encoded per dialect by + `Dialect.paramCodecs`. A per-driver codec axis can come later if a consumer needs one. +5. **Clauses (done, differently).** The plan was to move `compile.ts` into a + `ClickHouseDialect.buildSelect(state)`. The clauses Postgres needed to differ were few: + no `FORMAT`, an alias on a wrapped union, and GROUP BY by position (Postgres reads a bare + name as an input column before a select alias). Those are `Dialect.clauses` flags read at + three places in `compile.ts`, which kept ClickHouse output byte-identical without + relocating ~1100 lines. A dialect that needs a structurally different statement is the + point to revisit this. +6. **Postgres dialect (done).** `@maple-dev/effect-clickhouse/postgres`: `postgresDialect` + (double-quoted identifiers, standard strings with `E'…'` only to escape the param marker, + `$n` binding), column types, a function catalog (`count(*)`, `FILTER (WHERE …)`, + `percentile_cont`, `date_trunc(…, 'UTC')`, `date_bin`, `array_agg`, `->>`), and a + `compile` that defaults to Postgres. `src/pg/postgres.test.ts` runs every query on PGlite + (Postgres 17 in WASM), so a test passes only if Postgres accepts the SQL and the rows decode. ## Open questions - ClickHouse also supports server-side binding (`{name:Type}` with `query_params`). Adding it needs the param kind to name a ClickHouse type, which `param.of` already has. -- Whether the package splits (`core`, `clickhouse`, `postgres` entry points) at step 4 or 5. +- The package is still named for ClickHouse and the root entry still exports the ClickHouse + function catalog. Renaming, or a `core` entry without ClickHouse functions, is a packaging + decision for a later release. +- The ClickHouse function catalog writes ClickHouse SQL whatever the dialect. Functions could + refuse to compile under another dialect instead of producing SQL that fails at the server. +- MySQL or SQLite would need `?` placeholders (`reuse: false` already exists) and their own + types and functions; nothing in the core should need to change. diff --git a/docs/README.md b/docs/README.md index 74b1f50..1589c5e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -52,6 +52,7 @@ Roughly in reading order. | [Agent benchmark playbook](./benchmark-agent.md) | Repeatable optimization workflow and evidence checklist | | [Tenant scoping](./tenant-scoping.md) | `tenantScope`, what marks a query scoped, `crossTenant()` | | [Extending the DSL](./extending.md) | `defineFn`, raw escape hatches, handwritten SQL | +| [Postgres](./postgres.md) | The Postgres dialect, its column types and functions | ## Reference diff --git a/docs/params-and-compilation.md b/docs/params-and-compilation.md index 92c1c7b..cdfb8b2 100644 --- a/docs/params-and-compilation.md +++ b/docs/params-and-compilation.md @@ -189,13 +189,16 @@ Use `decodeRows` to validate wire values against the row schema. ## Dialects -A `Dialect` decides how literals are written and how resolved params reach the server. Its -`quoteString` and `literal` write every string fragment, every value compared against a column, -and every inline param, for the length of the compile. The default, `clickhouseDialect`, -writes each value into the SQL as a ClickHouse literal and leaves -`parameters` empty. A dialect whose `params` style is `bind` leaves a placeholder instead and -returns the encoded values in `parameters`, numbered once across the whole statement, unions -and subqueries included: +A `Dialect` is the database a query is compiled for: how identifiers and literals are written, +how resolved params reach the server, and which clauses exist. It is installed for the length of +the compile, so every column reference, string fragment, compared value and inline param goes +through it. The default, `clickhouseDialect`, writes names bare and each param value into the SQL +as a ClickHouse literal, and leaves `parameters` empty. `postgresDialect`, from the `/postgres` +entry point, is the other built-in one; see [Postgres](./postgres.md). + +A dialect whose `params` style is `bind` leaves a placeholder instead and returns the encoded +values in `parameters`, numbered once across the whole statement, unions and subqueries +included: ```ts const numbered: CH.Dialect = { @@ -212,15 +215,25 @@ const compiled = CH.compileUnsafe(query, { orgId: "org_1" }, { dialect: numbered `reuse: true` lets one numbered placeholder stand for every use of a param; set it to `false` for positional `?` placeholders, which bind a value each time they appear. Either way a bound value is the column codec's wire form, the same value an inline literal is written from, and a -missing or ill-typed param still fails the compile. +missing or ill-typed param still fails the compile. `paramCodecs` re-encodes a portable param +kind where the database wants another form: Postgres binds `param.bool` as a boolean rather +than `1`/`0`. + +| Member | Purpose | +| --- | --- | +| `quoteIdent` | One identifier: a column, alias, table or schema name | +| `quoteString`, `literal` | A string, or any encoded wire value, as a literal | +| `dateTimeLiteral` | A point in time compared against an expression with no declared type | +| `params` | `inline`, or `bind` with a placeholder function | +| `clauses.format` | Whether `FORMAT` exists; `.format()` fails to compile where it does not | +| `clauses.derivedTableAlias` | Whether a subquery in FROM needs an alias | +| `clauses.groupByAlias` | Whether GROUP BY resolves select aliases; if not, keys are written by position | +| `paramCodecs` | Per-kind codec overrides for `param.*` | Params are resolved by rewriting placeholders in the finished SQL, so a dialect's literals must never spell the param marker `__PARAM_`. ClickHouse writes it as `\x5F_PARAM_`; a literal that does contain the marker fails the compile with `InvalidLiteral` rather than being rewritten. -Identifiers, clauses, and functions are still ClickHouse SQL; a dialect changes literals and -param binding today. - ## Handwritten SQL When you need SQL the builder cannot express, `rawCompiledQuery` wraps a string in the same diff --git a/docs/postgres.md b/docs/postgres.md new file mode 100644 index 0000000..bf208b5 --- /dev/null +++ b/docs/postgres.md @@ -0,0 +1,120 @@ +# Postgres + +The same builder writes Postgres SQL. Queries, params, tenant scoping and row decoding work +as they do for ClickHouse; the `/postgres` entry point supplies the parts that differ: column +types whose codecs read what Postgres drivers send, functions spelled the Postgres way, and a +`compile` that defaults to the Postgres dialect. + +```ts title="postgres-quickstart.ts" +import { PGlite } from "@electric-sql/pglite" +import { Effect } from "effect" +import * as CH from "@maple-dev/effect-clickhouse" +import * as PG from "@maple-dev/effect-clickhouse/postgres" + +const Requests = CH.table( + "requests", + { OrgId: PG.text, Route: PG.text, DurationMs: PG.int8, At: PG.timestamptz }, + { tenantColumn: "OrgId" }, +) + +const query = CH.from(Requests) + .select(($) => ({ + route: $.Route, + count: PG.count(), + slow: PG.countIf($.DurationMs.gte(500)), + p50: PG.percentileCont(0.5, $.DurationMs), + })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId")), $.At.gte(CH.param.dateTime("since"))]) + .groupBy("route") + .orderBy(["count", "desc"]) + +export const compiled = PG.compileUnsafe(query, { orgId: "org_1", since: new Date("2026-01-01T00:00:00Z") }) +// compiled.sql: ... WHERE "requests"."OrgId" = $1 AND "requests"."At" >= $2 ... +// compiled.parameters: ["org_1", "2026-01-01T00:00:00.000Z"] + +const db = new PGlite() +await db.exec(` + CREATE TABLE requests ("OrgId" text, "Route" text, "DurationMs" int8, "At" timestamptz); + INSERT INTO requests VALUES + ('org_1', '/checkout', 120, '2026-01-01T10:00:00Z'), + ('org_1', '/checkout', 900, '2026-01-01T10:01:00Z'), + ('org_1', '/search', 40, '2026-01-01T10:02:00Z'); +`) +const result = await db.query>(compiled.sql, [...compiled.parameters]) +export const rows = await Effect.runPromise(compiled.decodeRows(result.rows)) +// [{ route: "/checkout", count: 2, slow: 1, p50: 510 }, { route: "/search", count: 1, slow: 0, p50: 40 }] +await db.close() +``` + +`PGlite` stands in for any driver that takes `(sql, values)`: node-postgres, postgres.js, or +`@effect/sql-pg`'s `unsafe`. The builder never runs the query. + +## What the dialect changes + +| | ClickHouse (`clickhouseDialect`) | Postgres (`postgresDialect`) | +| --- | --- | --- | +| Identifiers | Bare: `events.OrgId` | Quoted: `"events"."OrgId"` | +| String literals | Backslash escapes: `'it\'s'` | Doubled quotes: `'it''s'` | +| Params | Written in as literals; `parameters` empty | Bound as `$1`, `$2`, … in `parameters` | +| `param.bool` | `1` / `0` | `true` / `false` | +| `param.dateTime` | `'2026-01-01 00:00:00'` (UTC, zoneless) | `'2026-01-01T00:00:00.000Z'` | +| `GROUP BY` keys | Select aliases | Select-list positions (`GROUP BY 1`) | +| Wrapped union | `SELECT * FROM (…)` | `SELECT * FROM (…) AS "__union"` | +| `.format()` | `FORMAT JSON` | Refused at compile time | + +Postgres reads a bare name in `GROUP BY` as an input column before a select alias, so +`select({ Service: lower($.Service) }).groupBy("Service")` would group by the raw column. Writing +the position instead keeps ClickHouse's meaning. + +Every string that reaches the SQL as a literal is escaped for Postgres. A value that spells the +param marker `__PARAM_` is written as an `E'…'` string with the marker hex-escaped, and a +literal that still contained it would fail the compile with `InvalidLiteral`. + +## Column types + +| Constructor | Postgres type | Decodes to | Accepts on the wire | +| --- | --- | --- | --- | +| `text`, `uuid` | `text`, `uuid` | `string` | string | +| `bool` | `boolean` | `boolean` | boolean | +| `int2`, `int4`, `int8`, `float4`, `float8`, `numeric` | same | `number` | number, numeric string, `bigint` | +| `timestamptz` | `timestamptz` | `DateTime.Utc` | `Date`, or text such as `2026-01-01 00:00:00+00` | +| `jsonb(schema?)` | `jsonb` | the schema's type (`unknown` by default) | a parsed value | +| `array(type)` | `type[]` | `ReadonlyArray` | array | +| `nullable(type)` | the same type | `T \| null` | the same, or `null` | +| `custom(sql, schema, literalSchema?)` | anything | the schema's type | whatever the schema reads | + +`int8` and `numeric` decode to `number`, so values beyond 2^53 or a double's precision lose +digits. Declare `PG.custom("int8", Schema.String)` where exact digits matter. A `timestamptz` +compared against a `Date`, a `DateTime.Utc` or a string is written as an ISO-8601 instant, +which no session time zone can reinterpret; a zoneless string is read as UTC. + +## Functions + +| Function | SQL | Notes | +| --- | --- | --- | +| `count()`, `countDistinct(x)` | `count(*)`, `count(DISTINCT x)` | | +| `countIf(c)`, `sumIf(x, c)` | `count(*) FILTER (WHERE c)`, `sum(x) FILTER (WHERE c)` | ClickHouse's `-If` combinators | +| `sum`, `avg`, `min`, `max` | same | `null` over no rows, where ClickHouse returns `0` for `sum` | +| `percentileCont(f, x)` | `percentile_cont(f) WITHIN GROUP (ORDER BY x)` | Interpolated | +| `arrayAgg(x)` | `array_agg(x)` | | +| `dateTrunc(unit, ts)` | `date_trunc(unit, ts, 'UTC')` | UTC buckets; Postgres 12+ | +| `dateBin(seconds, ts)` | `date_bin(…, ts, epoch)` | Epoch-aligned buckets; Postgres 14+ | +| `now()` | `now()` | | +| `lower`, `upper`, `length` | same | | +| `coalesce(x, fallback)` | `coalesce(x, fallback)` | No longer nullable | +| `jsonText(x, key)` | `(x ->> key)` | `null` when absent | + +The shared operators (`eq`, `in_`, `like`, `ilike`, `and`, `or`, `not`, arithmetic, `lit`) work +unchanged. The ClickHouse function catalog on the root entry (`quantile`, `toStartOfInterval`, +map subscripts, …) writes ClickHouse SQL and will not run on Postgres. + +## Known differences + +- Integer division truncates in Postgres (`7 / 2` is `3`) and promotes to a float in ClickHouse. +- `rawExpr`, `untypedExpr`, `dynamicColumn` and `outerRef` take SQL as written; quote + identifiers yourself (`"OrgId"`) when targeting Postgres. +- An outer join fills missing columns with `NULL` in Postgres and with type defaults in + ClickHouse (unless `join_use_nulls` is set). Joined columns are typed nullable either way. + +See [`design/dialects.md`](../design/dialects.md) for how dialects are built and what a new one +needs to provide. diff --git a/docs/reference.md b/docs/reference.md index d767975..5243090 100644 --- a/docs/reference.md +++ b/docs/reference.md @@ -36,6 +36,7 @@ The root barrel is curated. These are exported by the package but not from it: | `Suite`, `RunOutput`, `BenchmarkError` and benchmark contracts | `/benchmark` | | `makeHttpClient`, `makeHttpTransport`, `httpConfigFromEnv` | `/benchmark/http` | | `runCli` | `/benchmark/cli` | +| `postgresDialect`, Postgres column types and functions, Postgres-default `compile` | `/postgres` | Every column-type constructor and every expression helper is on the root as well as on its subpath. See [Running a query](./running-queries.md) for what the `/sql` statement helpers are @@ -90,8 +91,10 @@ Note `/sql` exports a `compile` (fragment → string) distinct from the root `co | `rawCompiledQuery` | `({ sql, tenantScope, reason, justification, rowSchema?, route? }) => CompiledQuery` | | `clickhouseDialect` | The default `Dialect`: params written into the SQL as ClickHouse literals | -`Dialect` and `ParamStyle` describe how params reach the server; pass one as -`options.dialect`. See [Params and compilation](./params-and-compilation.md#dialects). +`Dialect`, `DialectClauses` and `ParamStyle` describe a database: how identifiers and literals +are written, how params reach the server, and which clauses exist. Pass one as +`options.dialect`. See [Params and compilation](./params-and-compilation.md#dialects) and +[Postgres](./postgres.md). ### Params @@ -273,7 +276,7 @@ Types: `WindowSpec`, `CompiledWindowSpec`, `WindowFrameBound`, `WindowRowsFrame` **Everything else** — `Table`, `TableOptions`, `Expr`, `ColumnRef`, `Condition`, `Comparable` (what a value of a type may be compared against), `MapValueOf`, `Subquery`, `ParamMarker`, `ParamKind`, `CHQuery`, `CHUnionQuery`, `ColumnAccessor`, `JoinedColumnAccessor`, -`JoinOnCallback`, `CompiledQuery`, `CompiledQueryInput`, `CompiledQueryRowSchema`, `RowSchemaMismatch`, `TenantScope`, `Dialect`, `ParamStyle`, `FnResult`, +`JoinOnCallback`, `CompiledQuery`, `CompiledQueryInput`, `CompiledQueryRowSchema`, `RowSchemaMismatch`, `TenantScope`, `Dialect`, `DialectClauses`, `ParamStyle`, `FnResult`, `WindowFunnelMode`, `WindowSpec`, `WindowRowsFrame`, `WindowFrameBound`, `WindowOrderDirection`, `CompiledWindowSpec`. diff --git a/package.json b/package.json index 828356d..286c2f6 100644 --- a/package.json +++ b/package.json @@ -46,6 +46,10 @@ "types": "./dist/sql.d.mts", "import": "./dist/sql.mjs" }, + "./postgres": { + "types": "./dist/postgres.d.mts", + "import": "./dist/postgres.mjs" + }, "./benchmark": { "types": "./dist/benchmark/index.d.mts", "import": "./dist/benchmark/index.mjs" @@ -77,6 +81,7 @@ "@effect/platform-bun": "4.0.0", "@effect/sql-clickhouse": "4.0.0", "@effect/vitest": "4.0.0", + "@electric-sql/pglite": "0.3.15", "@types/node": "^22.10.2", "effect": "4.0.0", "expect-type": "^1.3.0", diff --git a/scripts/check-doc-examples.mjs b/scripts/check-doc-examples.mjs index cb872e1..1a4fe03 100644 --- a/scripts/check-doc-examples.mjs +++ b/scripts/check-doc-examples.mjs @@ -101,6 +101,13 @@ const escapedSql = await import("./escaped-sql") const fragments = await import("@maple-dev/effect-clickhouse/sql") assert.equal(escapedSql.predicate, "Name = " + fragments.compile(fragments.str("O'Reilly"))) assert.doesNotMatch(escapedSql.predicate, /\\[object Object\\]/) +const postgres = await import("./postgres-quickstart") +assert.deepEqual(postgres.compiled.parameters, ["org_1", "2026-01-01T00:00:00.000Z"]) +assert.ok(sql(postgres.compiled).includes('"requests"."OrgId" = $1 AND "requests"."At" >= $2')) +assert.deepEqual(postgres.rows, [ + { route: "/checkout", count: 2, slow: 1, p50: 510 }, + { route: "/search", count: 1, slow: 0, p50: 40 }, +]) console.log("Markdown example behavior checks passed") `, ) diff --git a/src/ch/compile.ts b/src/ch/compile.ts index 4708634..01e4f7d 100644 --- a/src/ch/compile.ts +++ b/src/ch/compile.ts @@ -12,7 +12,7 @@ import type { CHQuery, CHQueryState } from "./query" import type { CHUnionQuery } from "./union" import { createQualifiedColumnAccessor, createJoinedColumnAccessor, sourceAlias } from "./query" import { aliased, columnTypeOf } from "./expr" -import { raw, ident, compile as compileSqlFragment } from "../sql/sql-fragment" +import { raw, identPath, quoteIdent, quoteIdentPath, compile as compileSqlFragment } from "../sql/sql-fragment" import { splitTerminalClauses } from "../sql/terminal-clauses" import { compileQuery, type SqlQuery } from "../sql/sql-query" import { PARAM_MARKER_PREFIX, PARAM_PLACEHOLDER_PATTERN, paramSchema, type ParamKind } from "./param" @@ -65,9 +65,34 @@ const orderByClause = (specs: ReadonlyArray<[string, "asc" | "desc"]>): Array): string => { + const position = selected.indexOf(key) + return currentDialect().clauses.groupByAlias || position === -1 ? quoteIdent(key) : String(position + 1) +} + +/** `.format()`'s value, refused for a dialect that has no `FORMAT` clause: + * dropping it would hand the caller rows in a shape they did not ask for. A + * defect, because the format is written in the query definition. */ +const formatClause = (format: string | undefined): string | undefined => { + if (format === undefined) return undefined + const dialect = currentDialect() + if (!dialect.clauses.format) { + throw new QueryBuilderDefect({ + message: `CHQuery: format(${JSON.stringify(format)}) has no meaning for the ${dialect.name} dialect, which has no FORMAT clause`, + }) + } + return format +} + // CompiledQuery — bundles the SQL string with its output type so consumers // never need to cast manually. @@ -654,16 +679,17 @@ function compileInner< enclosingCtes: visibleCtes, }) fromSource = sourceOf(inner) - fromFragment = raw(`(${inner.sql}) AS ${state.fromQueryAlias}`) + fromFragment = raw(`(${inner.sql}) AS ${quoteIdent(state.fromQueryAlias ?? "")}`) } else if (state.fromUnion) { const inner = compileUnionInner(state.fromUnion, params, { deferParams, nested: true, enclosingCtes: visibleCtes }) fromSource = sourceOf(inner) - fromFragment = raw(`(\n${splitTerminalClauses(inner.sql).body}\n) AS ${state.fromQueryAlias}`) + const body = currentDialect().clauses.format ? splitTerminalClauses(inner.sql).body : inner.sql + fromFragment = raw(`(\n${body}\n) AS ${quoteIdent(state.fromQueryAlias ?? "")}`) } else { fromSource = sourceForTable(state.tableName, mainColumn) fromFragment = mainAlias !== state.tableName - ? raw(`${state.tableName} AS ${mainAlias}`) - : ident(state.tableName) + ? raw(`${quoteIdentPath(state.tableName)} AS ${quoteIdentPath(mainAlias)}`) + : identPath(state.tableName) } const sources: TenantSource[] = [fromSource] @@ -696,7 +722,7 @@ function compileInner< tableSql = `(${compiled.sql})` source = sourceOf(compiled) } else if (j.tableName) { - tableSql = j.tableName + tableSql = quoteIdentPath(j.tableName) source = sourceForTable( j.tableName, j.tenantColumn === undefined ? undefined : `${j.alias}.${j.tenantColumn}`, @@ -722,7 +748,7 @@ function compileInner< return { type: j.type, table: tableSql, - alias: j.alias, + alias: quoteIdent(j.alias), on: on ? compileSqlFragment(on.toFragment()) : undefined, } }) @@ -732,7 +758,7 @@ function compileInner< from: fromFragment, joins, where: whereFragments, - groupBy: state.groupByKeys.map((k) => raw(k)), + groupBy: state.groupByKeys.map((k) => raw(groupByKey(k, options?.selectKeys ?? keys))), // Deliberately excluded from tenant evidence: by HAVING time the // rows are already aggregated, so the scan that produced them crossed // tenants no matter what this filters out. @@ -742,7 +768,7 @@ function compileInner< orderBy: orderByClause(state.orderBySpecs).map(raw), limit: state.limitValue != null ? raw(String(Math.round(state.limitValue))) : undefined, offset: state.offsetValue != null ? raw(String(Math.round(state.offsetValue))) : undefined, - format: options?.skipFormat ? undefined : state.formatValue, + format: options?.skipFormat ? undefined : formatClause(state.formatValue), } return compileQuery(sqlQuery) @@ -750,7 +776,7 @@ function compileInner< // Prepend CTE definitions if (resolvedCtes.length > 0) { - const cteDefs = resolvedCtes.map((c) => `${c.name} AS (\n${c.sql}\n)`).join(",\n") + const cteDefs = resolvedCtes.map((c) => `${quoteIdent(c.name)} AS (\n${c.sql}\n)`).join(",\n") sql = `WITH ${cteDefs}\n${sql}` } @@ -1067,7 +1093,7 @@ function compileUnionInner, Params extends Re state.outerOrderBySpecs.length > 0 || state.outerLimitValue != null || state.outerOffsetValue != null if (hasOuter) { - sql = `SELECT * FROM (\n${sql}\n)` + sql = `SELECT * FROM (\n${sql}\n)${currentDialect().clauses.derivedTableAlias ? ` AS ${quoteIdent("__union")}` : ""}` if (state.outerOrderBySpecs.length > 0) { sql += `\nORDER BY ${orderByClause(state.outerOrderBySpecs).join(", ")}` } @@ -1079,8 +1105,9 @@ function compileUnionInner, Params extends Re } } - if (state.formatValue) { - sql += `\nFORMAT ${state.formatValue}` + const format = formatClause(state.formatValue) + if (format) { + sql += `\nFORMAT ${format}` } let parameters: ReadonlyArray = [] @@ -1150,7 +1177,7 @@ function renderParams( missing.push(name) return placeholder } - const value = encodeParam(kind as ParamKind, name, params[name]) + const value = encodeParam(dialect, kind as ParamKind, name, params[name]) if (style._tag === "inline") return checkedLiteral(dialect, value, paramContext(kind, name)) const key = `${kind}\0${name}` @@ -1203,8 +1230,8 @@ const paramContext = (kind: string, name: string): string => `param '${name}' ($ * the two directions cannot drift: a `DateTime` param and a `DateTime` column * agree on the literal by construction, not by two functions being kept in sync. */ -function encodeParam(kind: ParamKind, name: string, value: unknown): unknown { - const schema = paramSchema(kind) +function encodeParam(dialect: Dialect, kind: ParamKind, name: string, value: unknown): unknown { + const schema = dialect.paramCodecs?.[kind] ?? paramSchema(kind) if (schema === undefined) { // Only reachable from a hand-written placeholder naming a kind nothing // declared: `param.of` registers its type before it can reach any SQL — diff --git a/src/ch/dialect.test.ts b/src/ch/dialect.test.ts index 842f59b..73ef480 100644 --- a/src/ch/dialect.test.ts +++ b/src/ch/dialect.test.ts @@ -19,6 +19,7 @@ const positional: Dialect = { // Standard SQL strings: a quote is doubled, a backslash is literal text. const standardQuote = (value: string) => `'${value.replace(/'/g, "''")}'` const literalDialect = (name: string, quoteString: (value: string) => string): Dialect => ({ + ...CH.clickhouseDialect, name, quoteString, literal: (value, context) => { @@ -182,3 +183,40 @@ describe("dialect literal syntax", () => { ) }) }) + +describe("dialect identifiers and clauses", () => { + const quoted: Dialect = { + ...CH.clickhouseDialect, + name: "quoted", + quoteIdent: (name) => `"${name.replace(/"/g, '""')}"`, + clauses: { format: false, derivedTableAlias: true, groupByAlias: true }, + } + const services = CH.table("db.services", { OrgId: CH.string, Service: CH.string }, { tenantColumn: "OrgId" }) + + it("quotes columns, qualifiers, tables, aliases, group and order keys", () => { + const query = CH.from(events) + .innerJoin(services, "s", (main, s) => main.Service.eq(s.Service)) + .select(($) => ({ service: $.s.Service, count: CH.count() })) + .where(($) => [$.OrgId.eq("o")]) + .groupBy("service") + .orderBy(["count", "desc"]) + const { sql } = compileCHUnsafe(query, {}, { dialect: quoted }) + expect(sql).toContain(`"s"."Service" AS "service"`) + expect(sql).toContain(`INNER JOIN "db"."services" AS "s" ON "events"."Service" = "s"."Service"`) + expect(sql).toContain(`FROM "events"`) + expect(sql).toContain(`WHERE "events"."OrgId" = 'o'`) + expect(sql).toContain(`GROUP BY "service"`) + expect(sql).toContain(`ORDER BY "count" DESC`) + }) + + it("aliases a wrapped union and refuses FORMAT where the dialect has neither", () => { + const branch = CH.from(events) + .select(($) => ({ count: $.Count })) + .where(($) => [$.OrgId.eq("o")]) + const union = compileUnionUnsafe(CH.unionAll(branch, branch).orderBy(["count", "asc"]), {}, { dialect: quoted }) + expect(union.sql).toContain(`) AS "__union"\nORDER BY "count" ASC`) + + expect(() => compileCHUnsafe(branch.format("JSON"), {}, { dialect: quoted })).toThrow(/no FORMAT clause/) + expect(compileCHUnsafe(branch.format("JSON"), {}).sql).toMatch(/FORMAT JSON$/) + }) +}) diff --git a/src/ch/dialect.ts b/src/ch/dialect.ts index 28523d0..7bc190c 100644 --- a/src/ch/dialect.ts +++ b/src/ch/dialect.ts @@ -2,16 +2,17 @@ // // What a compiled query needs to know about the database it is written for. // The builder, the tenant analysis, and row decoding are the same for every -// database; a dialect is the part that is not: how literals are written and how -// resolved param values reach the server. Identifier quoting, clause rendering, -// and column wire formats move behind it in later steps (see -// `design/dialects.md`). +// database; a dialect is the part that is not: how identifiers and literals are +// written, how resolved param values reach the server, which clauses exist, and +// how a portable param kind encodes. See `design/dialects.md`. +import type { Schema } from "effect" import { quoteClickHouseString } from "../sql/sql-fragment" -import { activeLiteralSyntax, withLiteralSyntax, type LiteralSyntax } from "../sql/literal-syntax" +import { activeSqlSyntax, withSqlSyntax, type SqlSyntax } from "../sql/sql-syntax" import { QueryBuilderError } from "./errors" import { sqlLiteral } from "./literal" import { PARAM_MARKER_PREFIX } from "./param" +import { chDateTimeLiteral } from "./types" /** * How resolved param values reach the server. @@ -38,6 +39,18 @@ export type ParamStyle = readonly reuse: boolean } +/** Clauses that exist in some dialects and not others. */ +export interface DialectClauses { + /** `FORMAT ` after a statement. A query with `.format()` fails to + * compile for a dialect without it. */ + readonly format: boolean + /** Whether a subquery in FROM needs an alias (`SELECT * FROM (…) AS u`). */ + readonly derivedTableAlias: boolean + /** Whether GROUP BY resolves a select alias before an input column of the + * same name. Where it does not, keys are written by select-list position. */ + readonly groupByAlias: boolean +} + /** * A database the builder writes SQL for. * @@ -46,36 +59,47 @@ export type ParamStyle = * finished text, so a value that spelled one would be rewritten too. A literal * that does is refused at compile time rather than trusted. */ -export interface Dialect extends LiteralSyntax { +export interface Dialect extends SqlSyntax { /** Shown in errors. */ readonly name: string readonly params: ParamStyle + readonly clauses: DialectClauses + /** + * The codec a portable param kind encodes with here, where it differs from + * the ClickHouse one: `param.bool` is `1`/`0` for ClickHouse and a boolean + * for Postgres, `param.dateTime` a zoneless string for one and an ISO-8601 + * instant for the other. Kinds not listed use the ClickHouse codec. + */ + readonly paramCodecs?: Readonly>> } /** ClickHouse, with params written into the SQL as literals. The default. */ export const clickhouseDialect: Dialect = { name: "clickhouse", + quoteIdent: (name) => name, quoteString: quoteClickHouseString, literal: sqlLiteral, + dateTimeLiteral: (value) => quoteClickHouseString(chDateTimeLiteral(value)), params: { _tag: "inline" }, + clauses: { format: true, derivedTableAlias: false, groupByAlias: true }, } // The dialect of the enclosing compile, beside the syntax installed for the -// fragment renderer. Same save/restore discipline as `withLiteralSyntax`. +// fragment renderer. Same save/restore discipline as `withSqlSyntax`. let current: Dialect | undefined /** The dialect of the enclosing compile, or ClickHouse outside one. */ export const currentDialect = (): Dialect => current ?? clickhouseDialect -/** Run `body` with `dialect`'s literal syntax installed, checked as above. */ +/** Run `body` with `dialect`'s syntax installed, literals checked as above. */ export function withDialect(dialect: Dialect, body: () => A): A { // A nested compile for the dialect already installed keeps the checked // syntax that is there instead of wrapping it again. - if (current === dialect && activeLiteralSyntax() !== undefined) return body() + if (current === dialect && activeSqlSyntax() !== undefined) return body() const previous = current current = dialect try { - return withLiteralSyntax(checkedSyntax(dialect), body) + return withSqlSyntax(checkedSyntax(dialect), body) } finally { current = previous } @@ -86,9 +110,11 @@ export function withDialect(dialect: Dialect, body: () => A): A { export const checkedLiteral = (dialect: Dialect, value: unknown, context: string): string => checked(dialect, dialect.literal(value, context), context) -const checkedSyntax = (dialect: Dialect): LiteralSyntax => ({ +const checkedSyntax = (dialect: Dialect): SqlSyntax => ({ + quoteIdent: dialect.quoteIdent, quoteString: (value) => checked(dialect, dialect.quoteString(value), "a string literal"), literal: (value, context) => checkedLiteral(dialect, value, context), + dateTimeLiteral: (value) => checked(dialect, dialect.dateTimeLiteral(value), "a DateTime literal"), }) const checked = (dialect: Dialect, sql: string, context: string): string => { diff --git a/src/ch/expr.ts b/src/ch/expr.ts index 4e5099b..3f8e658 100644 --- a/src/ch/expr.ts +++ b/src/ch/expr.ts @@ -8,7 +8,8 @@ import { DateTime, Result, Schema } from "effect" import type { SqlFragment } from "../sql/sql-fragment" -import { raw, str, compile, as_ as sqlAs, lazy } from "../sql/sql-fragment" +import { raw, str, ident, compile, as_ as sqlAs, lazy } from "../sql/sql-fragment" +import { activeSqlSyntax } from "../sql/sql-syntax" import { chDateTimeLiteral, CHFloatResult, CHNumber, string as chString, type CHType, type InferTS } from "./types" import { encodeColumnLiteral } from "./literal" import { markTenantColumn, markTenantPredicate, tenantColumnOf, tenantPredicatesOf } from "./tenant" @@ -155,14 +156,24 @@ export function toFragment(value: unknown): SqlFragment { if (isExprLike(value)) return value.toFragment() if (typeof value === "string") return str(value) if (typeof value === "number") return raw(String(value)) - if (typeof value === "boolean") return raw(value ? "1" : "0") + if (typeof value === "boolean") return lazy(() => untypedLiteral(value)) // A DateTime column compares against a DateTime value, so the literal has to - // be ClickHouse's tz-less form rather than whatever `String(value)` produces. - if (DateTime.isDateTime(value)) return str(chDateTimeLiteral(DateTime.toUtc(value))) - if (value instanceof Date) return str(chDateTimeLiteral(DateTime.makeUnsafe(value))) + // be the dialect's own form (ClickHouse's is tz-less) rather than whatever + // `String(value)` produces. + if (DateTime.isDateTime(value)) return lazy(() => dateTimeLiteral(DateTime.toUtc(value))) + if (value instanceof Date) return lazy(() => dateTimeLiteral(DateTime.makeUnsafe(value))) return raw(String(value)) } +/** A value with no column type to encode it, in the active dialect's syntax. + * ClickHouse writes booleans as `1`/`0`, which is also the rendering outside + * a compile. */ +const untypedLiteral = (value: boolean): string => + activeSqlSyntax()?.literal(value, "an untyped boolean") ?? (value ? "1" : "0") + +const dateTimeLiteral = (value: DateTime.Utc): string => + activeSqlSyntax()?.dateTimeLiteral(value) ?? compile(str(chDateTimeLiteral(value))) + // Expr implementation /** Whether a codec accepts `null` — asked, not inferred from its AST, so it @@ -312,7 +323,10 @@ export function makeColumnRef { - const fragment = raw(name) + // `alias.Column` when qualified: the qualifier is quoted segment by segment, + // the column as one identifier (a ClickHouse `Nested` column has a dot). + const qualified = columnName !== undefined && name.endsWith(`.${columnName}`) + const fragment = qualified ? ident(columnName, name.slice(0, -columnName.length - 1)) : ident(name) const base = makeExpr>( fragment, columnType?.schema as Schema.Codec, any> | undefined, @@ -366,7 +380,7 @@ export function makeColumnRef { - return makeExpr(lazy(() => `${name}[${compile(str(key))}]`), columnType?.element?.schema) + return makeExpr(lazy(() => `${compile(fragment)}[${compile(str(key))}]`), columnType?.element?.schema) }, }, ) as ColumnRef diff --git a/src/ch/index.ts b/src/ch/index.ts index c66e169..7875d01 100644 --- a/src/ch/index.ts +++ b/src/ch/index.ts @@ -258,7 +258,7 @@ export { } from "./compile" // Dialects: how a compiled query's params reach the server. -export { clickhouseDialect, type Dialect, type ParamStyle } from "./dialect" +export { clickhouseDialect, type Dialect, type DialectClauses, type ParamStyle } from "./dialect" // Failures vs defects — the rule the two classes encode is on `QueryBuilderError`. export { QueryBuilderError, QueryBuilderDefect } from "./errors" diff --git a/src/ch/literal.ts b/src/ch/literal.ts index 09a5612..8ec6dbf 100644 --- a/src/ch/literal.ts +++ b/src/ch/literal.ts @@ -14,7 +14,7 @@ import { Result, Schema } from "effect" import { QueryBuilderError } from "./errors" import { quoteClickHouseString } from "../sql/sql-fragment" -import { activeLiteralSyntax } from "../sql/literal-syntax" +import { activeSqlSyntax } from "../sql/sql-syntax" import type { CHType } from "./types" const isPlainObject = (value: unknown): value is Record => @@ -89,7 +89,7 @@ export function sqlLiteral(value: unknown, context: string): string { * fails here — while building the SQL — instead of becoming part of it. */ export function encodeLiteral(schema: Schema.Codec, value: unknown, context: string): string { - return (activeLiteralSyntax()?.literal ?? sqlLiteral)(encodeValue(schema, value, context), context) + return (activeSqlSyntax()?.literal ?? sqlLiteral)(encodeValue(schema, value, context), context) } /** diff --git a/src/pg/dialect.ts b/src/pg/dialect.ts new file mode 100644 index 0000000..67217ef --- /dev/null +++ b/src/pg/dialect.ts @@ -0,0 +1,86 @@ +// The Postgres dialect. + +import { DateTime, Schema } from "effect" +import type { Dialect } from "../ch/dialect" +import { QueryBuilderError } from "../ch/errors" +import { PARAM_MARKER_PREFIX } from "../ch/param" +import { PgTimestampLiteral, timestampLiteral } from "./types" + +const quoteIdent = (name: string): string => `"${name.replace(/"/g, '""')}"` + +/** + * A string literal. Plain `'...'` with quotes doubled, which reads backslashes + * as text (`standard_conforming_strings`, the default since Postgres 9.1). A + * value that contains the param marker is written as an `E'...'` string instead, + * the one form where the marker can be hex-escaped (`\x5F` is `_`). + */ +const quoteString = (value: string): string => { + if (value.includes("\0")) { + throw new QueryBuilderError({ + code: "InvalidLiteral", + message: "a string literal: Postgres text cannot contain a NUL character", + }) + } + if (!value.includes(PARAM_MARKER_PREFIX)) return `'${value.replace(/'/g, "''")}'` + const escaped = value + .replace(/\\/g, "\\\\") + .replace(/'/g, "''") + .replace(/__PARAM_/g, "\\x5F_PARAM_") + return `E'${escaped}'` +} + +const isPlainObject = (value: unknown): value is Record => + typeof value === "object" && + value !== null && + (Object.getPrototypeOf(value) === Object.prototype || Object.getPrototypeOf(value) === null) + +/** An encoded wire value as a Postgres literal. */ +const literal = (value: unknown, context: string): string => { + if (value === null) return "NULL" + if (typeof value === "string") return quoteString(value) + if (typeof value === "boolean") return value ? "TRUE" : "FALSE" + if (typeof value === "bigint") return String(value) + if (typeof value === "number") { + if (!Number.isFinite(value)) { + throw new QueryBuilderError({ code: "InvalidLiteral", message: `${context}: ${value} has no Postgres literal` }) + } + return String(value) + } + // `ARRAY[]` needs a cast to say its element type; the untyped `'{}'` takes + // the type of whatever it is compared against. + if (Array.isArray(value)) { + return value.length === 0 ? "'{}'" : `ARRAY[${value.map((element) => literal(element, context)).join(", ")}]` + } + // A record reaches here from a jsonb codec that did not stringify it. + if (isPlainObject(value)) return quoteString(JSON.stringify(value)) + throw new QueryBuilderError({ + code: "InvalidLiteral", + message: `${context}: cannot write ${typeof value} as a Postgres literal`, + }) +} + +/** `param.dateTimeSeconds`: the same instant, floored to whole seconds. */ +const timestampSeconds = timestampLiteral((epochMillis) => new Date(Math.floor(epochMillis / 1000) * 1000).toISOString()) + +/** + * Postgres: double-quoted identifiers, standard string literals, and params + * bound to `$1`, `$2`, … and returned in `CompiledQuery.parameters`. + * + * Portable param kinds encode for Postgres: `param.bool` binds a boolean + * rather than ClickHouse's `1`/`0`, and `param.dateTime` an ISO-8601 instant + * rather than a zoneless string a session time zone could reinterpret. + */ +export const postgresDialect: Dialect = { + name: "postgres", + quoteIdent, + quoteString, + literal, + dateTimeLiteral: (value) => `TIMESTAMPTZ ${quoteString(DateTime.formatIso(value))}`, + params: { _tag: "bind", placeholder: (index) => `$${index}`, reuse: true }, + clauses: { format: false, derivedTableAlias: true, groupByAlias: false }, + paramCodecs: { + bool: Schema.Boolean, + dateTime: PgTimestampLiteral, + dateTimeSeconds: timestampSeconds, + }, +} diff --git a/src/pg/functions.ts b/src/pg/functions.ts new file mode 100644 index 0000000..4b13813 --- /dev/null +++ b/src/pg/functions.ts @@ -0,0 +1,120 @@ +// Postgres functions. +// +// The shared operators (`eq`, `and`, `in_`, `like`, arithmetic, `not`) come +// from the builder and render the same everywhere. These are the ones whose +// Postgres spelling or result type differs from ClickHouse's: `count(*)` rather +// than `count()`, `FILTER (WHERE …)` rather than `-If` combinators, and +// aggregates over no rows returning NULL rather than 0. + +import { Schema, type DateTime } from "effect" +import { QueryBuilderDefect } from "../ch/errors" +import { makeExpr, type Condition, type Expr } from "../ch/expr" +import { schemaOf, withoutNull } from "../ch/define-fn" +import { compile, lazy, raw, str } from "../sql/sql-fragment" +import * as T from "./types" + +const sql = (expr: Expr | Condition): string => compile(expr.toFragment()) + +const nullableNumber = Schema.NullOr(T.PgNumber) as Schema.Codec +const int8 = T.int8.schema as Schema.Codec + +// Aggregates + +/** `count(*)`. */ +export const count = (): Expr => makeExpr(raw("count(*)"), int8) + +/** `count(DISTINCT expr)`. */ +export const countDistinct = (expr: Expr): Expr => + makeExpr(lazy(() => `count(DISTINCT ${sql(expr)})`), int8) + +/** `count(*) FILTER (WHERE condition)`: ClickHouse's `countIf`. */ +export const countIf = (condition: Condition): Expr => + makeExpr(lazy(() => `count(*) FILTER (WHERE ${sql(condition)})`), int8) + +/** `sum(expr)`. NULL over no rows, and a string for int8/numeric inputs on the + * wire, which the result codec reads as a number. */ +export const sum = (expr: Expr): Expr => + makeExpr(lazy(() => `sum(${sql(expr)})`), nullableNumber) + +/** `sum(expr) FILTER (WHERE condition)`: ClickHouse's `sumIf`. */ +export const sumIf = (expr: Expr, condition: Condition): Expr => + makeExpr(lazy(() => `sum(${sql(expr)}) FILTER (WHERE ${sql(condition)})`), nullableNumber) + +/** `avg(expr)`. NULL over no rows. */ +export const avg = (expr: Expr): Expr => + makeExpr(lazy(() => `avg(${sql(expr)})`), nullableNumber) + +const nullableOf = (expr: Expr): Schema.Codec | undefined => { + const schema = schemaOf(expr) + return schema === undefined ? undefined : (Schema.NullOr(schema) as Schema.Codec) +} + +/** `min(expr)`, decoding as `expr` does. NULL over no rows. */ +export const min = (expr: Expr): Expr => makeExpr(lazy(() => `min(${sql(expr)})`), nullableOf(expr)) + +/** `max(expr)`, decoding as `expr` does. NULL over no rows. */ +export const max = (expr: Expr): Expr => makeExpr(lazy(() => `max(${sql(expr)})`), nullableOf(expr)) + +/** `percentile_cont(fraction) WITHIN GROUP (ORDER BY expr)`: an interpolated + * quantile, ClickHouse's `quantileExact` family. */ +export const percentileCont = (fraction: number, expr: Expr): Expr => { + if (!(fraction >= 0 && fraction <= 1)) { + throw new QueryBuilderDefect({ message: `percentileCont: fraction must be within [0, 1], got ${fraction}` }) + } + return makeExpr(lazy(() => `percentile_cont(${fraction}) WITHIN GROUP (ORDER BY ${sql(expr)})`), nullableNumber) +} + +/** `array_agg(expr)`. NULL over no rows. */ +export const arrayAgg = (expr: Expr): Expr | null> => { + const element = schemaOf(expr) + return makeExpr( + lazy(() => `array_agg(${sql(expr)})`), + element === undefined ? undefined : (Schema.NullOr(Schema.Array(element)) as Schema.Codec | null, unknown>), + ) +} + +// Time + +const timestamptz = T.timestamptz.schema as Schema.Codec + +export type DateTruncUnit = "second" | "minute" | "hour" | "day" | "week" | "month" | "quarter" | "year" + +/** `date_trunc(unit, ts, 'UTC')`: buckets in UTC whatever the session time + * zone, as ClickHouse's `toStartOf*` functions do. Postgres 12+. */ +export const dateTrunc = (unit: DateTruncUnit, ts: Expr): Expr => + makeExpr(lazy(() => `date_trunc(${compile(str(unit))}, ${sql(ts)}, 'UTC')`), timestamptz) + +/** `date_bin(seconds, ts, epoch)`: fixed-width buckets aligned to the Unix + * epoch, ClickHouse's `toStartOfInterval`. Postgres 14+. */ +export const dateBin = (seconds: number, ts: Expr): Expr => { + if (!(Number.isSafeInteger(seconds) && seconds > 0)) { + throw new QueryBuilderDefect({ message: `dateBin: bucket width must be a positive whole number of seconds, got ${seconds}` }) + } + return makeExpr( + lazy(() => `date_bin(make_interval(secs => ${seconds}), ${sql(ts)}, TIMESTAMPTZ '1970-01-01 00:00:00+00')`), + timestamptz, + ) +} + +/** `now()`: the transaction's start time. */ +export const now = (): Expr => makeExpr(raw("now()"), timestamptz) + +// Strings and values + +const text = T.text.schema as Schema.Codec + +export const lower = (expr: Expr): Expr => makeExpr(lazy(() => `lower(${sql(expr)})`), text) +export const upper = (expr: Expr): Expr => makeExpr(lazy(() => `upper(${sql(expr)})`), text) +export const length = (expr: Expr): Expr => + makeExpr(lazy(() => `length(${sql(expr)})`), T.int4.schema as Schema.Codec) + +/** `coalesce(expr, fallback)`, no longer nullable. */ +export const coalesce = (expr: Expr, fallback: Expr): Expr => + makeExpr( + lazy(() => `coalesce(${sql(expr)}, ${sql(fallback)})`), + schemaOf(fallback) ?? withoutNull(schemaOf(expr)), + ) + +/** `expr ->> key`: a jsonb field as text, NULL when it is absent. */ +export const jsonText = (expr: Expr, key: string): Expr => + makeExpr(lazy(() => `(${sql(expr)} ->> ${compile(str(key))})`), Schema.NullOr(Schema.String) as Schema.Codec) diff --git a/src/pg/postgres.test.ts b/src/pg/postgres.test.ts new file mode 100644 index 0000000..1dd604d --- /dev/null +++ b/src/pg/postgres.test.ts @@ -0,0 +1,216 @@ +// The Postgres dialect against a real Postgres: PGlite (Postgres 17 compiled to +// WASM), in-process. Every query here is compiled by the builder and executed, +// so a test passes only if Postgres parses the SQL, binds the params, and the +// rows decode through the declared types. + +import { PGlite } from "@electric-sql/pglite" +import { afterAll, beforeAll, describe, expect, it } from "@effect/vitest" +import { DateTime, Effect } from "effect" +import type { CompiledQuery } from "../ch/compile" +import * as CH from "../index" +import * as PG from "../postgres" + +const db = new PGlite() + +const events = CH.table( + "events", + { + OrgId: PG.text, + Service: PG.text, + Count: PG.int8, + Timestamp: PG.timestamptz, + Ok: PG.bool, + Attrs: PG.jsonb(), + }, + { tenantColumn: "OrgId" }, +) + +const services = CH.table("app.services", { OrgId: PG.text, Service: PG.text, Team: PG.nullable(PG.text) }, { tenantColumn: "OrgId" }) + +const tricky = "it's \\ a __PARAM_string_orgId__ value" + +beforeAll(async () => { + await db.exec(` + CREATE SCHEMA app; + CREATE TABLE events ("OrgId" text, "Service" text, "Count" int8, "Timestamp" timestamptz, "Ok" boolean, "Attrs" jsonb); + CREATE TABLE app.services ("OrgId" text, "Service" text, "Team" text); + INSERT INTO events VALUES + ('org_1', 'api', 10, '2026-01-01T00:00:10Z', true, '{"region": "eu"}'), + ('org_1', 'api', 20, '2026-01-01T00:04:59Z', false, '{"region": "eu"}'), + ('org_1', 'web', 5, '2026-01-01T00:05:00Z', true, '{"region": "us"}'), + ('org_1', $$${tricky}$$, 1, '2026-01-01T01:00:00Z', true, '{}'), + ('org_2', 'api', 99, '2026-01-01T00:00:00Z', true, '{"region": "eu"}'); + INSERT INTO app.services VALUES ('org_1', 'api', 'core'), ('org_1', 'web', NULL); + `) +}) + +afterAll(async () => { + await db.close() +}) + +const run = async (compiled: CompiledQuery): Promise> => { + const result = await db.query>(compiled.sql, [...compiled.parameters]) + return Effect.runPromise(compiled.decodeRows(result.rows)) +} + +describe("postgres dialect", () => { + it("aggregates per group with bound params, FILTER, and quoted identifiers", async () => { + const query = CH.from(events) + .select(($) => ({ + service: $.Service, + events: PG.count(), + failures: PG.countIf($.Ok.eq(false)), + total: PG.sum($.Count), + average: PG.avg($.Count), + })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId")), $.Service.in_("api", "web")]) + .groupBy("service") + .orderBy(["service", "asc"]) + const compiled = PG.compileUnsafe(query, { orgId: "org_1" }) + + expect(compiled.sql).toContain(`"OrgId" = $1`) + expect(compiled.parameters).toEqual(["org_1"]) + expect(compiled.tenantScope).toBe("single-tenant") + expect(await run(compiled)).toEqual([ + { service: "api", events: 2, failures: 1, total: 30, average: 15 }, + { service: "web", events: 1, failures: 0, total: 5, average: 5 }, + ]) + }) + + it("decodes timestamptz and buckets with dateBin and dateTrunc in UTC", async () => { + const query = CH.from(events) + .select(($) => ({ + bucket: PG.dateBin(300, $.Timestamp), + hour: PG.dateTrunc("hour", $.Timestamp), + events: PG.count(), + })) + .where(($) => [ + $.OrgId.eq("org_1"), + $.Timestamp.gte(CH.param.dateTime("start")), + $.Timestamp.lt(CH.param.dateTime("end")), + ]) + .groupBy("bucket", "hour") + .orderBy(["bucket", "asc"]) + const rows = await run( + PG.compileUnsafe(query, { + start: new Date("2026-01-01T00:00:00Z"), + end: DateTime.makeUnsafe("2026-01-01T00:30:00Z"), + }), + ) + expect(rows.map((row) => [DateTime.formatIso(row.bucket), DateTime.formatIso(row.hour), row.events])).toEqual([ + ["2026-01-01T00:00:00.000Z", "2026-01-01T00:00:00.000Z", 2], + ["2026-01-01T00:05:00.000Z", "2026-01-01T00:00:00.000Z", 1], + ]) + }) + + it("joins a schema-qualified table, reads a subquery, and nulls a missing join value", async () => { + const perService = CH.from(events) + .select(($) => ({ OrgId: $.OrgId, Service: $.Service, total: PG.sum($.Count) })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId"))]) + .groupBy("OrgId", "Service") + const query = CH.fromQuery(perService, "per") + .innerJoin(services, "s", (per, s) => per.Service.eq(s.Service).and(per.OrgId.eq(s.OrgId))) + .select(($) => ({ service: $.Service, team: PG.coalesce($.s.Team, CH.lit("none")), total: $.total })) + .orderBy(["service", "asc"]) + const compiled = PG.compileUnsafe(query, { orgId: "org_1" }) + expect(compiled.sql).toContain(`INNER JOIN "app"."services" AS "s"`) + expect(await run(compiled)).toEqual([ + { service: "api", team: "core", total: 30 }, + { service: "web", team: "none", total: 5 }, + ]) + }) + + it("runs a CTE and an ordered union, which Postgres needs aliased", async () => { + const recent = CH.table("recent", { OrgId: PG.text, Count: PG.int8 }) + const cte = CH.from(recent) + .withCTE( + "recent", + CH.from(events) + .select(($) => ({ OrgId: $.OrgId, Count: $.Count })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId"))]), + ) + .select(($) => ({ total: PG.sum($.Count) })) + expect(await run(PG.compileUnsafe(cte, { orgId: "org_1" }))).toEqual([{ total: 36 }]) + + const branch = (org: string) => + CH.from(events) + .select(($) => ({ org: $.OrgId, total: PG.sum($.Count) })) + .where(($) => [$.OrgId.eq(CH.param.string(org))]) + .groupBy("org") + const union = PG.compileUnionUnsafe( + CH.unionAll(branch("first"), branch("second")).orderBy(["total", "desc"]), + { first: "org_1", second: "org_2" }, + ) + expect(union.parameters).toEqual(["org_1", "org_2"]) + expect(await run(union)).toEqual([ + { org: "org_2", total: 99 }, + { org: "org_1", total: 36 }, + ]) + }) + + // The value spells a placeholder and carries a quote and a backslash, and is + // matched three ways: as a bound param, as a column literal, and through + // LIKE. Each has to reach Postgres as data, not as SQL or as a placeholder. + it("keeps quotes, backslashes and the param marker inside values", async () => { + const byParam = CH.from(events) + .select(($) => ({ count: $.Count })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId")), $.Service.eq(CH.param.string("service"))]) + expect(await run(PG.compileUnsafe(byParam, { orgId: "org_1", service: tricky }))).toEqual([{ count: 1 }]) + + const byLiteral = CH.from(events) + .select(($) => ({ count: $.Count })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId")), $.Service.eq(tricky), $.Service.like("it's \\\\%")]) + const compiled = PG.compileUnsafe(byLiteral, { orgId: "org_1" }) + expect(compiled.sql).toContain(`E'it''s \\\\ a \\x5F_PARAM_string_orgId__ value'`) + expect(compiled.parameters).toEqual(["org_1"]) + expect(await run(compiled)).toEqual([{ count: 1 }]) + }) + + it("binds booleans and compares jsonb", async () => { + const query = CH.from(events) + .select(($) => ({ region: PG.jsonText($.Attrs, "region"), regions: PG.arrayAgg($.Service) })) + .where(($) => [ + $.OrgId.eq("org_1"), + $.Ok.eq(CH.param.bool("ok")), + $.Attrs.neq({ region: "us" }), + ]) + .groupBy("region") + .orderBy(["region", "asc"]) + const compiled = PG.compileUnsafe(query, { ok: true }) + expect(compiled.parameters).toEqual([true]) + expect(await run(compiled)).toEqual([ + { region: "eu", regions: ["api"] }, + { region: null, regions: [tricky] }, + ]) + }) + + it("computes an interpolated percentile", async () => { + const query = CH.from(events) + .select(($) => ({ p50: PG.percentileCont(0.5, $.Count), lowest: PG.min($.Count) })) + .where(($) => [$.OrgId.eq("org_1")]) + expect(await run(PG.compileUnsafe(query, {}))).toEqual([{ p50: 7.5, lowest: 1 }]) + }) + + it("refuses FORMAT and fails a missing param before reaching Postgres", () => { + const query = CH.from(events) + .select(($) => ({ count: $.Count })) + .where(($) => [$.OrgId.eq(CH.param.string("orgId"))]) + expect(() => PG.compileUnsafe(query.format("JSON"), { orgId: "o" })).toThrow(/no FORMAT clause/) + expect(() => PG.compileUnsafe(query, {})).toThrow(/no value given for param 'orgId'/) + }) +}) + +describe("postgres group by", () => { + // An alias that shadows an input column: Postgres would group a bare + // `GROUP BY "Service"` by the raw column, splitting 'api' and 'API'. + it("groups by the selected expression, not a same-named input column", async () => { + await db.exec(`INSERT INTO events VALUES ('org_3', 'API', 1, now(), true, '{}'), ('org_3', 'api', 2, now(), true, '{}')`) + const query = CH.from(events) + .select(($) => ({ Service: PG.lower($.Service), total: PG.sum($.Count) })) + .where(($) => [$.OrgId.eq("org_3")]) + .groupBy("Service") + const compiled = PG.compileUnsafe(query, {}) + expect(compiled.sql).toContain("GROUP BY 1") + expect(await run(compiled)).toEqual([{ Service: "api", total: 3 }]) + }) +}) diff --git a/src/pg/types.ts b/src/pg/types.ts new file mode 100644 index 0000000..f5a6cca --- /dev/null +++ b/src/pg/types.ts @@ -0,0 +1,156 @@ +// Postgres Column Types +// +// The same descriptor as the ClickHouse types (`CHType`): a SQL type name, the +// codec its wire value decodes with, and the codec a compared literal encodes +// with. Only the codecs differ. Postgres drivers disagree on wire forms (int8 and +// numeric arrive as strings from node-postgres and postgres.js, timestamptz as a +// `Date` or a string depending on parser settings), so each codec accepts every +// form a common driver sends, the way `CHNumber` accepts a quoted 64-bit integer. + +import { DateTime, Schema, SchemaGetter } from "effect" +import { chDateTimeToIso, custom, type CHType, type InferEncoded, type InferTS } from "../ch/types" + +/** A Postgres column type. The ClickHouse descriptor under a dialect-neutral name. */ +export type PgType = CHType + +/** A number as any Postgres driver sends one: a number, a numeric string, or a + * `bigint`. Decodes to `number`, so an int8 beyond 2^53 loses precision; declare + * `custom("int8", Schema.String)` where that matters. */ +export const PgNumber: Schema.Codec = Schema.Union([ + Schema.Finite, + Schema.FiniteFromString, + Schema.BigInt.pipe( + Schema.decodeTo(Schema.Finite, { + decode: SchemaGetter.transform((value: bigint) => Number(value)), + encode: SchemaGetter.transform((value: number) => globalThis.BigInt(value)), + }), + ), +]) + +/** + * Postgres's text form of a timestamptz (`2026-01-01 00:00:00.25+00`) as + * ISO-8601, which `Date` parses the same way on every runtime. A zoneless value + * is read as UTC, the same rule the ClickHouse types apply. + */ +export const pgTimestampToIso = (value: string): string => { + const trimmed = value.trim() + const zoned = /^(\d{4}-\d{2}-\d{2})[ T](\d{2}:\d{2}:\d{2}(?:\.\d+)?)([+-]\d{2})(?::?(\d{2}))?$/.exec(trimmed) + if (zoned) return `${zoned[1]}T${zoned[2]}${zoned[3]}:${zoned[4] ?? "00"}` + return chDateTimeToIso(trimmed) +} + +const isTimestamp = Schema.makeFilter( + (value: string) => + Number.isNaN(Date.parse(pgTimestampToIso(value))) + ? `\`${value}\` is not a timestamp: expected ISO-8601 or \`YYYY-MM-DD hh:mm:ss[+zz]\`` + : undefined, + { title: "postgresTimestamp" }, +) + +/** A timestamptz from its text form. */ +const TimestampFromString: Schema.Codec = Schema.String.pipe( + Schema.check(isTimestamp), + Schema.decodeTo(Schema.DateTimeUtc, { + decode: SchemaGetter.transform((value: string) => DateTime.makeUnsafe(pgTimestampToIso(value))), + encode: SchemaGetter.transform((value: DateTime.Utc) => DateTime.formatIso(value)), + }), +) + +/** A timestamptz as a driver sends it: a `Date` (the usual parse) or text. */ +const PgTimestamptz: Schema.Codec = Schema.Union([ + Schema.DateTimeUtcFromDate, + TimestampFromString, +]) + +/** + * Everything a timestamptz can be compared against (a `DateTime.Utc`, a `Date`, + * or a timestamp string), written as an ISO-8601 instant by `format`, which + * receives epoch milliseconds. Unlike a zoneless literal, an instant cannot be + * read in the session's time zone. + */ +export const timestampLiteral = (format: (epochMillis: number) => string): Schema.Codec => + Schema.Union([ + Schema.String.pipe( + Schema.decodeTo(Schema.DateTimeUtc, { + decode: SchemaGetter.transform((value: string) => DateTime.makeUnsafe(pgTimestampToIso(value))), + encode: SchemaGetter.transform((value: DateTime.Utc) => format(DateTime.toEpochMillis(value))), + }), + ), + Schema.String.pipe( + Schema.decodeTo(Schema.Date, { + decode: SchemaGetter.transform((value: string) => new Date(value)), + encode: SchemaGetter.transform((value: Date) => format(value.getTime())), + }), + ), + Schema.String.pipe( + Schema.check(isTimestamp), + Schema.decodeTo(Schema.String, { + decode: SchemaGetter.transform((value: string) => value), + encode: SchemaGetter.transform((value: string) => format(Date.parse(pgTimestampToIso(value)))), + }), + ), + ]) as Schema.Codec + +const isoInstant = (epochMillis: number): string => new Date(epochMillis).toISOString() + +export const PgTimestampLiteral: Schema.Codec = timestampLiteral(isoInstant) + +// Scalars + +export const text: PgType<"text", string> = custom("text", Schema.String) +export const uuid: PgType<"uuid", string> = custom("uuid", Schema.String) +export const bool: PgType<"boolean", boolean> = custom("boolean", Schema.Boolean) +export const int2: PgType<"int2", number, number | string | bigint> = custom("int2", PgNumber) +export const int4: PgType<"int4", number, number | string | bigint> = custom("int4", PgNumber) +export const int8: PgType<"int8", number, number | string | bigint> = custom("int8", PgNumber) +export const float4: PgType<"float4", number, number | string | bigint> = custom("float4", PgNumber) +export const float8: PgType<"float8", number, number | string | bigint> = custom("float8", PgNumber) +/** Decodes to `number`; precision beyond a double is lost. */ +export const numeric: PgType<"numeric", number, number | string | bigint> = custom("numeric", PgNumber) +export const timestamptz: PgType<"timestamptz", DateTime.Utc, Date | string> = custom( + "timestamptz", + PgTimestamptz, + PgTimestampLiteral, +) + +/** + * A jsonb column, decoded with `schema` (anything, by default). Drivers parse + * jsonb themselves, so rows arrive as values; a compared literal is written as + * JSON text, which Postgres reads back as jsonb. + */ +export const jsonb = ( + schema: Schema.Codec = Schema.Unknown as Schema.Codec, +): PgType<"jsonb", A, unknown> => + custom("jsonb", schema as Schema.Codec, Schema.fromJsonString(schema) as Schema.Codec) + +// Compound types + +export type PgArray> = PgType< + "Array", + ReadonlyArray>, + ReadonlyArray> +> + +export type PgNullable> = PgType< + "Nullable", + InferTS | null, + InferEncoded | null +> + +export const array = >(e: E): PgArray => ({ + _tag: "Array", + sql: `${e.sql}[]`, + schema: Schema.Array(e.schema), + literalSchema: Schema.Array(e.literalSchema) as Schema.Codec, + element: e, +}) as PgArray + +export const nullable = >(t: T): PgNullable => ({ + _tag: "Nullable", + sql: t.sql, + schema: Schema.NullOr(t.schema), + literalSchema: Schema.NullOr(t.literalSchema) as Schema.Codec, + element: t, +}) as PgNullable + +export { custom } diff --git a/src/postgres.ts b/src/postgres.ts new file mode 100644 index 0000000..66c151f --- /dev/null +++ b/src/postgres.ts @@ -0,0 +1,34 @@ +// @maple-dev/effect-clickhouse/postgres +// +// Postgres for the same query builder. Build queries with the root entry +// (`table`, `from`, `param`, `unionAll`, the shared operators); declare +// columns with the types here and compile with the `compile` here, which +// defaults to the Postgres dialect. See docs/postgres.md. + +import { + compileCH, + compileCHUnsafe, + compileUnion as compileUnionCH, + compileUnionUnsafe as compileUnionUnsafeCH, +} from "./ch/compile" +import { postgresDialect } from "./pg/dialect" + +export { postgresDialect } +export * from "./pg/types" +export * from "./pg/functions" + +/** `compile` from the root entry, for Postgres unless `options.dialect` says otherwise. */ +export const compile: typeof compileCH = (query, params, options) => + compileCH(query, params, { ...options, dialect: options?.dialect ?? postgresDialect }) + +/** `compileUnsafe` from the root entry, for Postgres. */ +export const compileUnsafe: typeof compileCHUnsafe = (query, params, options) => + compileCHUnsafe(query, params, { ...options, dialect: options?.dialect ?? postgresDialect }) + +/** `compileUnion` from the root entry, for Postgres. */ +export const compileUnion: typeof compileUnionCH = (union, params, options) => + compileUnionCH(union, params, { ...options, dialect: options?.dialect ?? postgresDialect }) + +/** `compileUnionUnsafe` from the root entry, for Postgres. */ +export const compileUnionUnsafe: typeof compileUnionUnsafeCH = (union, params, options) => + compileUnionUnsafeCH(union, params, { ...options, dialect: options?.dialect ?? postgresDialect }) diff --git a/src/sql/literal-syntax.ts b/src/sql/literal-syntax.ts deleted file mode 100644 index c714a2c..0000000 --- a/src/sql/literal-syntax.ts +++ /dev/null @@ -1,32 +0,0 @@ -// Literal syntax of the dialect being compiled for. -// -// Literals are written while a query's callbacks run, deep inside expression -// code that has no dialect argument to read. `compile` installs the dialect's -// syntax here for the duration of the compile instead, the same way -// `withSubqueryCompiler` hands down the subquery renderer. A leaf module so the -// fragment renderer can read it without importing the builder. - -export interface LiteralSyntax { - /** A string as a quoted literal. */ - readonly quoteString: (value: string) => string - /** An encoded wire value (string, number, boolean, null, arrays and records - * of those) as a literal. Throws for a value it cannot write. */ - readonly literal: (value: unknown, context: string) => string -} - -// Compilation is synchronous. Save/restore makes nested compilations -// independent, including when a callback throws. -let current: LiteralSyntax | undefined - -export function withLiteralSyntax(syntax: LiteralSyntax, body: () => A): A { - const previous = current - current = syntax - try { - return body() - } finally { - current = previous - } -} - -/** The syntax installed by the enclosing compile, if there is one. */ -export const activeLiteralSyntax = (): LiteralSyntax | undefined => current diff --git a/src/sql/sql-fragment.ts b/src/sql/sql-fragment.ts index 366e8ce..3217776 100644 --- a/src/sql/sql-fragment.ts +++ b/src/sql/sql-fragment.ts @@ -1,5 +1,5 @@ import { Data } from "effect" -import { activeLiteralSyntax } from "./literal-syntax" +import { activeSqlSyntax } from "./sql-syntax" // ClickHouse string escaping @@ -33,7 +33,14 @@ export const quoteClickHouseString = (value: string): string => `'${escapeClickH /** A string literal in the syntax of the dialect being compiled for, or * ClickHouse's outside a compile. */ -const quoteString = (value: string): string => (activeLiteralSyntax()?.quoteString ?? quoteClickHouseString)(value) +const quoteString = (value: string): string => (activeSqlSyntax()?.quoteString ?? quoteClickHouseString)(value) + +/** One identifier in the syntax of the dialect being compiled for. ClickHouse + * names are written bare, which is also the rendering outside a compile. */ +export const quoteIdent = (name: string): string => activeSqlSyntax()?.quoteIdent(name) ?? name + +/** A dotted path (`db.table`, `alias.Column`) with each segment quoted. */ +export const quoteIdentPath = (path: string): string => path.split(".").map(quoteIdent).join(".") // SQL Fragment AST @@ -44,8 +51,14 @@ export type SqlFragment = Data.TaggedEnum<{ Str: { readonly value: string } /** Integer parameter: produces the number as string, rounded */ Int: { readonly value: number } - /** Column or table identifier (unquoted — ClickHouse style) */ - Ident: { readonly name: string } + /** + * An identifier, quoted by the dialect being compiled for. + * + * `name` is one identifier and is never split, so a ClickHouse `Nested` + * column such as `Events.Name` stays whole. `qualifier` is the dotted path in + * front of it (`alias`, `db.table`), quoted segment by segment. + */ + Ident: { readonly name: string; readonly qualifier?: string } /** A list of fragments joined by a separator (empty strings from When(false) are filtered) */ Join: { readonly separator: string; readonly fragments: ReadonlyArray } /** An aliased expression: AS */ @@ -72,7 +85,13 @@ const Frag = Data.taggedEnum() export const raw = (sql: string): SqlFragment => Frag.Raw({ sql }) export const str = (value: string): SqlFragment => Frag.Str({ value }) export const int = (value: number): SqlFragment => Frag.Int({ value }) -export const ident = (name: string): SqlFragment => Frag.Ident({ name }) +export const ident = (name: string, qualifier?: string): SqlFragment => + Frag.Ident(qualifier === undefined ? { name } : { name, qualifier }) +/** A dotted path such as `db.table`, split into qualifier and name. */ +export const identPath = (path: string): SqlFragment => { + const dot = path.lastIndexOf(".") + return dot === -1 ? ident(path) : ident(path.slice(dot + 1), path.slice(0, dot)) +} export const join = (separator: string, ...fragments: ReadonlyArray): SqlFragment => Frag.Join({ separator, fragments }) export const as_ = (expr: SqlFragment, alias: string): SqlFragment => Frag.As({ expr, alias }) @@ -86,9 +105,10 @@ export const compile: (fragment: SqlFragment) => string = Frag.$match({ Raw: ({ sql }) => sql, Str: ({ value }) => quoteString(value), Int: ({ value }) => String(Math.round(value)), - Ident: ({ name }) => name, + Ident: ({ name, qualifier }) => + qualifier === undefined ? quoteIdent(name) : `${quoteIdentPath(qualifier)}.${quoteIdent(name)}`, Join: ({ separator, fragments }) => fragments.map(compile).filter(Boolean).join(separator), - As: ({ expr, alias }) => `${compile(expr)} AS ${alias}`, + As: ({ expr, alias }) => `${compile(expr)} AS ${quoteIdent(alias)}`, When: ({ condition, fragment }) => (condition ? compile(fragment) : ""), Lazy: ({ render }) => render(), }) diff --git a/src/sql/sql-syntax.ts b/src/sql/sql-syntax.ts new file mode 100644 index 0000000..a4e6af0 --- /dev/null +++ b/src/sql/sql-syntax.ts @@ -0,0 +1,38 @@ +// Syntax of the dialect being compiled for. +// +// Identifiers and literals are written while a query's callbacks run, deep +// inside expression code that has no dialect argument to read. `compile` +// installs the dialect's syntax here for the duration of the compile instead, +// the same way `withSubqueryCompiler` hands down the subquery renderer. A leaf +// module so the fragment renderer can read it without importing the builder. + +import type { DateTime } from "effect" + +export interface SqlSyntax { + /** One identifier (a column, alias, table or schema name), quoted as needed. */ + readonly quoteIdent: (name: string) => string + /** A string as a quoted literal. */ + readonly quoteString: (value: string) => string + /** An encoded wire value (string, number, boolean, null, arrays and records + * of those) as a literal. Throws for a value it cannot write. */ + readonly literal: (value: unknown, context: string) => string + /** A point in time compared against an expression with no declared type. */ + readonly dateTimeLiteral: (value: DateTime.Utc) => string +} + +// Compilation is synchronous. Save/restore makes nested compilations +// independent, including when a callback throws. +let current: SqlSyntax | undefined + +export function withSqlSyntax(syntax: SqlSyntax, body: () => A): A { + const previous = current + current = syntax + try { + return body() + } finally { + current = previous + } +} + +/** The syntax installed by the enclosing compile, if there is one. */ +export const activeSqlSyntax = (): SqlSyntax | undefined => current diff --git a/tsdown.config.ts b/tsdown.config.ts index f2f473c..edee4c9 100644 --- a/tsdown.config.ts +++ b/tsdown.config.ts @@ -6,6 +6,7 @@ export default defineConfig({ expr: "./src/expr.ts", types: "./src/types.ts", sql: "./src/sql/index.ts", + postgres: "./src/postgres.ts", "benchmark/index": "./src/benchmark/index.ts", "benchmark/http": "./src/benchmark/http.ts", "benchmark/cli": "./src/benchmark/cli.ts",