From 974850bb525a08ed15db94ee1bc49661cfb08a86 Mon Sep 17 00:00:00 2001 From: Makisuo Date: Sat, 3 Oct 2026 19:15:29 +0200 Subject: [PATCH 1/3] Add schema definitions and migrations for ClickHouse Three opt-in entry points: - ./schema: defineTable (a Table that carries its DDL), materializedView (its body is a DSL query, type-checked against the target table), DDL rendering with replicated engines and ON CLUSTER as render options, content-hashed snapshots, and an offline diff into migration ops. - ./kit and the effect-orm command: generate writes the next migration, asks before dropping data (or takes --hints and exits 2 without them), and refuses changes that need a table rebuild; check validates the snapshot chain and conflicting branches. - ./migrate: run, status, and verify through a MigrationDriver built from a SqlClient. Each statement is journaled after it finishes and the migration is recorded last, so a failed run resumes; applied migrations have their hash checked; verify compares the database with the snapshot of the last applied migration, normalized by the server. Plan and implementation notes in design/migrations.md, user docs in docs/migrations.md. --- CHANGELOG.md | 12 + design/migrations.md | 485 +++++++++++++++++++++++++++++++ docs/README.md | 7 +- docs/migrations.md | 171 +++++++++++ package.json | 15 +- scripts/check-package.ts | 1 + src/kit.ts | 19 ++ src/kit/bin.ts | 3 + src/kit/cli.ts | 158 ++++++++++ src/kit/generate.ts | 189 ++++++++++++ src/kit/graph.test.ts | 65 +++++ src/kit/graph.ts | 129 ++++++++ src/kit/kit.test.ts | 91 ++++++ src/migrate.ts | 29 ++ src/migrate/driver.ts | 67 +++++ src/migrate/errors.ts | 38 +++ src/migrate/ledger.ts | 105 +++++++ src/migrate/run.ts | 168 +++++++++++ src/migrate/source.ts | 174 +++++++++++ src/migrate/verify.ts | 208 +++++++++++++ src/schema.ts | 58 ++++ src/schema/define.ts | 364 +++++++++++++++++++++++ src/schema/diff.ts | 228 +++++++++++++++ src/schema/entities.ts | 132 +++++++++ src/schema/ops.ts | 135 +++++++++ src/schema/render.ts | 116 ++++++++ src/schema/schema.test.ts | 192 ++++++++++++ src/schema/snapshot.ts | 83 ++++++ tests/migrate.clickhouse.test.ts | 219 ++++++++++++++ tests/package-consumer.mts | 25 ++ tsdown.config.ts | 4 + 31 files changed, 3688 insertions(+), 2 deletions(-) create mode 100644 design/migrations.md create mode 100644 docs/migrations.md create mode 100644 src/kit.ts create mode 100644 src/kit/bin.ts create mode 100644 src/kit/cli.ts create mode 100644 src/kit/generate.ts create mode 100644 src/kit/graph.test.ts create mode 100644 src/kit/graph.ts create mode 100644 src/kit/kit.test.ts create mode 100644 src/migrate.ts create mode 100644 src/migrate/driver.ts create mode 100644 src/migrate/errors.ts create mode 100644 src/migrate/ledger.ts create mode 100644 src/migrate/run.ts create mode 100644 src/migrate/source.ts create mode 100644 src/migrate/verify.ts create mode 100644 src/schema.ts create mode 100644 src/schema/define.ts create mode 100644 src/schema/diff.ts create mode 100644 src/schema/entities.ts create mode 100644 src/schema/ops.ts create mode 100644 src/schema/render.ts create mode 100644 src/schema/schema.test.ts create mode 100644 src/schema/snapshot.ts create mode 100644 tests/migrate.clickhouse.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index db3aa37..b5476e3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,18 @@ ## Unreleased +- Add schema-as-code and migrations for ClickHouse, all opt-in (see `docs/migrations.md`): + - `./schema`: `defineTable` (a `Table` that also carries its DDL), `materializedView` (its + body is a DSL query, type-checked against the target table), DDL rendering with replicated + engines and `ON CLUSTER` as render options, content-hashed snapshots, and an offline diff. + - `./kit` and the `effect-orm` command: `generate` writes the next migration from the schema + modules, asks before dropping data (or takes `--hints`, exiting 2 without them), and + refuses changes that need a table rebuild; `check` validates the snapshot chain and + branch conflicts. + - `./migrate`: `run`, `status`, and `verify` through a `MigrationDriver` you build from + your `SqlClient`. Statements are journaled one by one so a failed run resumes, applied + migrations have their hash checked, and `verify` compares the database with the last + applied snapshot. - Rename the package to `@maple-dev/effect-orm` and the repository to `MapleTechLabs/effect-orm`. Imports, error `_tag` prefixes (`@maple-dev/effect-orm/QueryBuilderError`, ...) and the live-test variables (`EFFECT_ORM_CLICKHOUSE_URL`, `_USER`, `_PASSWORD`) change with it. diff --git a/design/migrations.md b/design/migrations.md new file mode 100644 index 0000000..20f7e75 --- /dev/null +++ b/design/migrations.md @@ -0,0 +1,485 @@ +# Migrations: schema-as-code, snapshots, and a migrator + +Status: phases 0 to 3 implemented on 2026-10-03 (branch `feat/migrations`); phases 4 to 7 open. +Section 7 lists what was built and where it departs from this plan. User docs: +[`docs/migrations.md`](../docs/migrations.md). + +Today `@maple-dev/effect-orm` models tables for **querying**: `table(name, columns)` is a name and a +column record, `docs/tables-and-types.md` calls it "a contract you keep in sync with your +migrations", and the package never touches the network. This doc studies how Drizzle does +migrations (drizzle-orm and drizzle-kit `1.0.0-rc.5-5935859`, read from source), notes Effect's +own `effect/sql/Migrator` (effect `4.0.0`), and proposes how effect-orm can own the schema too. + +**In one paragraph.** Add an opt-in DDL layer (`defineTable`, `materializedView`) whose values are +still ordinary `Table`s, so nothing about querying changes. Copy Drizzle's authoring model almost +whole: an offline `generate` that diffs a committed per-migration snapshot against the code, +rename resolution by prompt or hints, `check` for branch conflicts. Do **not** copy its runtime for +ClickHouse: no transactions, apply-by-name without hash checks, and "all pending in one batch" are +wrong there. Ship a migrator that runs on Effect's `SqlClient` (already part of the `effect` peer +dependency), journals per statement, verifies hashes, and has first-class ClickHouse operations +for what `ALTER` cannot do. ClickHouse first. Postgres second, because drizzle-kit already serves it. + +--- + +## 1. How Drizzle does it (v1 RC) + +### Schema to snapshot to SQL (`generate`) + +1. The TypeScript schema is serialized (`fromDrizzleSchema`, then `interimToDDL`) into a **flat + list of entities** tagged with `entityType` (`tables`, `columns`, `pks`, `fks`, `indexes`, ...). +2. The previous snapshot is the **last migration folder by name**. An empty folder diffs against a + "dry" snapshot. +3. `ddlDiff` diffs one entity kind at a time (schemas, enums, tables, columns, indexes, pks, fks, + views, ...). Renames resolved for a kind are applied to the old side before the next kind. +4. Every diff statement maps to exactly one convertor (`createTableConvertor`, + `addColumnConvertor`, ...); an unmapped one throws `No convertor for`. +5. Output: `drizzle/_/{migration.sql, snapshot.json}`, statements joined by + `--> statement-breakpoint`. No changes writes nothing. **`generate` never connects to a database.** + +```json +{ "id": "", "prevIds": [""], "version": "8", "dialect": "postgres", + "ddl": [{ "entityType": "tables", "name": "...", "schema": "public" }], "renames": [] } +``` + +### Ordering, identity, tracking table + +- v1 has **no journal**. Order is lexical by folder name, so the timestamp prefix is the order. +- `prevIds` is an array, so a merge snapshot can have two parents. +- Runtime (`drizzle-orm/migrator.js`): one entry per folder, `hash = sha256(migration.sql)`, SQL + split on the breakpoint. +- Table: `drizzle.__drizzle_migrations (id serial, hash text, created_at bigint, name text, applied_at timestamptz)`. + A v0 table (`id, hash, created_at`) is upgraded in place and its rows matched back to folders. +- **What runs is decided by name only**: every local folder not in the table. A branch migration + with an older timestamp still runs. **The stored hash is never checked again**, so editing an + applied migration goes unnoticed. +- All pending migrations run in **one transaction**. + +### Commands + +| Command | Behavior | +| ---------- | ------------------------------------------------------------------------------------------------------------------------------ | +| `generate` | Offline diff, writes a folder. `--custom` writes an empty SQL file with a snapshot copied from its parent, keeping the chain continuous. `--name` sets the slug. | +| `migrate` | `check`, then the runtime migrator. | +| `push` | Introspect the live DB, diff against the code, apply directly. No files, no history. | +| `pull` | Introspect into TypeScript plus a commented `migration.sql`. `--init` records it as applied (the baseline). | +| `check` | Validate snapshot versions and detect conflicting branches. | +| `up` | Upgrade folder layout and snapshot versions. | +| `drop` | Removed. Delete the folder by hand. | + +### Renames and ambiguity + +When one entity kind has both creates and deletes, a TTY gets a prompt: "Is X created or renamed +from another X?". Without a TTY drizzle-kit **does not guess**: the ambiguity becomes a missing +hint, nothing is written, and it exits 2. The caller retries with `--hints` or `--hints-file` +(`{type:"rename",kind,from,to}`, `{type:"create",...}`, `{type:"confirm_data_loss",...}`). +`--output json` makes it machine-readable, which is what makes it usable by agents and CI. + +### Rollback and branches + +- **No down migrations.** Rollback only happens when the transaction undoes a failed batch. +- `check` builds a parent-to-children graph from `prevIds`. Where a parent has several children it + diffs each branch and intersects their footprints. Overlap is a `conflicts` error. Disjoint + branches are "commutative", and the next `generate` writes a merge snapshot with both leaves as + `prevIds`. +- A real hazard, seen in a consumer repo: two folders created in the same second + (`20260706224607_electric_publication_wave1` and `20260706224607_huge_dexter_bennett`) apply in + name order even though the first one's `prevIds` points at the second. + +### The lesson + +Drizzle's value is in **authoring**: a snapshot per migration turns "what changed" into a pure, +offline diff, and `check` turns branch conflicts into a CI failure. Its **runtime** is deliberately +thin because Postgres DDL is transactional. ClickHouse DDL is not, so the runtime is where +effect-orm has to do more than Drizzle, not less. + +--- + +## 2. Effect's own `Migrator`, and why it is not enough for ClickHouse + +`effect/sql/Migrator` (re-exported as `@effect/sql-clickhouse/ClickhouseMigrator`) is hand-written +migrations, not schema-as-code: a loader returns `[id, name, Effect]` triples (`fromGlob`, +`fromRecord`, `fromFileSystem`) and the runner tracks `effect_sql_migrations (migration_id, name, created_at)`. +Reading `node_modules/effect/src/sql/Migrator.ts`: + +- It applies migrations with **`id > max(applied id)`**. A branch migration with a lower id that + merges later is silently skipped. Drizzle's name set does not have this problem. +- It **inserts the ledger rows first**, then runs the migrations, all inside + `sql.withTransaction`. The "lock" is a primary-key conflict on that insert. +- On ClickHouse, the client's `beginTransaction` is `BEGIN TRANSACTION` (experimental, MergeTree + only, needs a session), primary keys are not unique, and the ledger table is created by the + generic `orElse` branch. So the lock cannot work, and a migration that fails halfway is likely + **already recorded as applied**. Confirm this on a live server (phase 0); if it holds, it is + worth reporting upstream either way. +- No hashes, no snapshots, no diff. + +What we keep from it: migrations as Effects that need `SqlClient`, the loader shapes, and the fact +that `SqlClient` lives in the `effect` package. That means a migrator can ship in effect-orm +**without a new dependency and without the library opening a connection itself**: the caller +provides a `SqlClient` layer (`@effect/sql-clickhouse`, `@effect/sql-pg`, ...), exactly as they do +for queries today. + +--- + +## 3. What transfers, what does not + +### Transfers from Drizzle + +| Drizzle | effect-orm | +| --------------------------------------------------------- | -------------------------------------------------------------------------------------------- | +| Schema in TypeScript is the only source of truth | `defineTable` / `materializedView` values; still usable as `Table`s in queries | +| Flat, entity-tagged snapshot JSON per migration | Same idea, ClickHouse entity kinds (section 4.2) | +| Offline `generate`, `--custom`, `--name` | Same | +| Kind-by-kind diff, renames applied before the next kind | Same order, ClickHouse kinds | +| TTY prompts; non-TTY hints, exit 2, `--output json` | Same contract and hint shapes, so tooling written for drizzle-kit hints carries over | +| Folder per migration, timestamp prefix, `prevIds` DAG | Same layout; ordering fixed to follow the DAG (4.3) | +| `check` with commutativity | Same, with footprints per ClickHouse object | +| `pull --init` baseline | `pull` writes `defineTable` code from `system.*`; `--init` records the baseline | +| `push` for development | Same, refused unless `--dev` or a local URL | +| Name-based "what is pending" | Same, and **the hash is verified** (Drizzle stores it and never reads it) | + +### Does not transfer to ClickHouse + +- **Transactions.** There are none for DDL, and a migration can stop between statements. Instead: + generated statements are idempotent (`IF [NOT] EXISTS`), each statement is journaled after it + finishes, and a rerun resumes at the first unjournaled statement. The ledger row for the whole + migration is written **last**, never first. +- **Arbitrary `ALTER`.** You cannot change an engine, `PARTITION BY`, or the primary key, and + `MODIFY ORDER BY` can only append columns added in the same `ALTER`. These become a typed + **rebuild**: create `__new`, backfill in windows, `EXCHANGE TABLES` (Atomic database) or + rename, recreate dependent views, drop the old table. The generator emits the rebuild op and + refuses to emit an `ALTER` the server would reject. +- **Materialized views.** The SELECT is frozen at creation. A body change is `DROP VIEW` + + `CREATE MATERIALIZED VIEW` (inserts in the gap are not materialized) or `ALTER TABLE MODIFY QUERY` + where the server allows it. New views never use `POPULATE`; history comes from an explicit + backfill ordered so it and the live view never write the same rows. Drops are ordered + `DROP VIEW` before `DROP TABLE`; the inverse leaves a view pointing at a missing target and every + insert into its source fails with `UNKNOWN_TABLE`. +- **Mutations.** `MODIFY COLUMN` type changes, `MATERIALIZE COLUMN`, `MATERIALIZE INDEX` and + `ALTER ... UPDATE/DELETE` rewrite parts in the background. A step that starts one waits on + `system.mutations` (or runs with `mutations_sync = 2`) before it is journaled, and the plan + labels it "rewrites data". +- **TTL.** `MODIFY TTL` runs with `materialize_ttl_after_modify = 0` unless the migration opts in, + so a TTL edit does not rewrite a large table by accident. +- **Backfills.** A single `INSERT ... SELECT` over a big table can outlast any HTTP timeout. A + `backfill` op declares target, columns, source, time column and projection; the runner splits it + into day-aligned windows and journals each window. Convergence on rerun is the author's job + (truncate or rebuild the target first), and `check` lints for it. +- **Clusters.** Self-managed clusters may need `ON CLUSTER ` on every statement and + `Replicated*` engines, while ClickHouse Cloud uses `SharedMergeTree` and needs neither. Engine + flavor and cluster name are **render-time options**, not part of the schema, so one schema + serves all three. The ledger must be replicated too, or replicas disagree about what ran. +- **Locking.** No advisory locks. The runner takes a lease row (owner, expiry) in the ledger and + refuses to start while a live one exists. This is best effort and documented as such; callers + that need a hard guarantee serialize runs themselves. +- **Rollback.** None, as in Drizzle. Dropped data does not come back. Document + expand/contract: add in one release, stop reading the old shape, drop in a later one. +- **Server versions.** Some objects only exist on newer servers (text indexes, for example). A + schema entry can declare `minServerVersion`. The migrator skips it with a recorded reason on older + servers and installs it on the next run after an upgrade, without blocking the migrations that + correctness depends on. + +--- + +## 4. Design + +### 4.1 Schema definition (`@maple-dev/effect-orm/schema`) + +```ts +import * as S from "@maple-dev/effect-orm/schema" +import * as T from "@maple-dev/effect-orm/types" + +export const Spans = S.defineTable("spans", { + columns: { + OrgId: S.column(T.custom("LowCardinality(String)", Schema.String)), + Timestamp: S.column(T.dateTime64, { codec: "Delta, ZSTD(1)" }), + ServiceName: S.column(T.string), + Duration: S.column(T.uint64, { default: 0 }), + }, + engine: S.engine.mergeTree(), + orderBy: ["OrgId", "ServiceName", "Timestamp"], + partitionBy: ($) => CH.toDate($.Timestamp), + ttl: ($) => S.ttl.delete(CH.toDate($.Timestamp), { days: 30 }), + indexes: [S.index.bloomFilter("idx_trace", ($) => $.TraceId, { granularity: 1 })], + tenantColumn: "OrgId", +}) + +export const SpansHourly = S.materializedView("spans_hourly_mv", { + to: SpansHourlyTarget, + as: CH.from(Spans).select(($) => ({ OrgId: $.OrgId, Hour: CH.toStartOfHour($.Timestamp), Count: CH.count() })).groupBy("OrgId", "Hour"), +}) +``` + +- `defineTable` returns a value that **is** a `Table` (`_tag: "Table"`, same `name`, `columns` + readable as `ColumnDefs`, same `tenantColumn`), plus a `ddl` field. Every existing query API + accepts it unchanged; `table()` stays as it is for people who manage DDL elsewhere. +- Expressions in `partitionBy`, `ttl`, defaults, and index expressions reuse the query DSL, so + they are typechecked against the table's columns. +- A materialized view's body **is a query built with the DSL**, compiled with the ClickHouse + dialect at snapshot time. Its output row is checked against the target table's columns at the + type level, which catches the drift (a view writing a column the target lacks, or the wrong type) + that today only surfaces as a failed insert. +- `T.custom`'s SQL name is the column type in DDL. No new type system. + +### 4.2 Snapshot + +```json +{ + "version": "1", + "dialect": "clickhouse", + "id": "", + "prevIds": [""], + "entities": [ + { "kind": "table", "name": "spans", "engine": { "family": "MergeTree", "params": [] }, + "orderBy": "OrgId, ServiceName, Timestamp", "partitionBy": "toDate(Timestamp)", + "primaryKey": null, "ttl": "toDate(Timestamp) + toIntervalDay(30)", "settings": {} }, + { "kind": "column", "table": "spans", "name": "Duration", "position": 3, "type": "UInt64", + "default": { "kind": "DEFAULT", "expr": "0" }, "codec": null, "comment": null }, + { "kind": "index", "table": "spans", "name": "idx_trace", "expr": "TraceId", "type": "bloom_filter", "granularity": 1 }, + { "kind": "materialized_view", "name": "spans_hourly_mv", "to": "spans_hourly", + "sources": ["spans"], "select": "" } + ], + "renames": [] +} +``` + +- Flat, sorted, deterministic, like Drizzle's `ddl`. +- `id` is a **content hash**, not a random UUID: two branches that reach the same schema agree, and + `check` can recompute every id. +- SQL fragments (expressions, MV bodies) are stored in the server's canonical form. `generate` + stays offline, so canonicalization is the library's own printer; `verify` compares against the + live server with `formatQuery()` applied to both sides, so whitespace and quoting never show up + as drift. +- Postgres snapshots use the same envelope with Postgres kinds (`table`, `column`, `pk`, `fk`, + `unique`, `check`, `index`, `view`, `enum`). + +### 4.3 Migration folder and ordering + +``` +migrations/ + 20261003120000_init/ + snapshot.json + migration.sql # generated; breakpoint-separated + 20261005093000_add_service_version/ + snapshot.json + migration.sql + 20261007150000_backfill_hourly/ + snapshot.json # copied from parent (custom) + migration.ts # default export: a Plan of ops, or an Effect needing SqlClient +``` + +- Same layout as Drizzle v1, so people know it on sight. +- `migration.sql` covers plain DDL. `migration.ts` is for ops SQL cannot express (backfill + windows, rebuilds, waits on mutations) and for arbitrary Effects (Effect `Migrator` style). The + generator writes `.ts` automatically when a change needs a rebuild or backfill. +- **Order follows the `prevIds` DAG** (topological, folder name breaks ties between independent + branches). `check` fails when names and the DAG disagree, which turns the same-second hazard + into an error. +- Pending is decided **by name** (Drizzle), never by "greater than the last id" (Effect). + +### 4.4 Generator + +`effect-orm generate [--name x] [--custom] [--hints ...] [--output json]`: + +1. Load the config (`effect-orm.config.ts`: `dialect`, `schema` glob, `out`, render options). +2. Import the schema modules, collect `defineTable` and `materializedView` values, build the + snapshot. Offline. +3. Run `check`; diff against the DAG leaf (or the common ancestor with merged branch statements + replayed, as drizzle-kit does). +4. Diff kind by kind: tables, columns, indexes, settings, TTL, keys, then views. Resolve renames + before moving to the next kind. +5. Classify each change: + +| Change | Emitted | Plan label | +| ------------------------------------------------ | ------------------------------------------------------------------------------- | ------------------- | +| New table / view | `CREATE ... IF NOT EXISTS`, views without `POPULATE` | metadata | +| Add column | `ADD COLUMN IF NOT EXISTS ... AFTER ...`, then recreate views that should write it | metadata | +| Default, comment, codec | `MODIFY COLUMN` (codec affects new parts only) | metadata | +| Column type | `MODIFY COLUMN` + mutation wait; lossy casts need a hint | rewrites data | +| Skip index | `ADD/DROP INDEX IF [NOT] EXISTS`; `MATERIALIZE INDEX` only when asked | metadata / rewrite | +| TTL | `MODIFY TTL` with `materialize_ttl_after_modify = 0` | metadata | +| `ORDER BY` append of a column added in this change | one `ALTER` with `ADD COLUMN` and `MODIFY ORDER BY` | metadata | +| Engine, `PARTITION BY`, primary key, other key change | `rebuild` op in `migration.ts`, backfill projection derived from the column mapping | rebuild | +| View body | `DROP VIEW` + `CREATE MATERIALIZED VIEW` (`MODIFY QUERY` behind an option) | metadata, ingest gap | +| Drop column / table / view | dependents first, `DROP VIEW` before `DROP TABLE`; needs `confirm_data_loss` | destructive | +| Rename table / column | `RENAME TABLE` / `RENAME COLUMN IF EXISTS`, then recreate views that read it | metadata | + +6. Write the folder and print the plan grouped by label, so "rewrites data", "ingest gap" and + "destructive" lines stand out. + +### 4.5 Runtime (`@maple-dev/effect-orm/migrate`) + +```ts +const applied = yield* Migrate.run({ + loader: Migrate.fromFileSystem("./migrations"), // or fromRecord(import.meta.glob(...)) for bundlers + dialect: "clickhouse", + render: { engineFlavor: "Replicated", cluster: "main" }, + strict: true, +}) // Effect, MigrateError, SqlClient> +``` + +Ledger, ClickHouse: + +```sql +CREATE TABLE IF NOT EXISTS _effect_orm_migrations ( + name String, hash String, prev_ids Array(String), + applied_at DateTime64(3) DEFAULT now64(3), status LowCardinality(String) +) ENGINE = ReplacingMergeTree(applied_at) ORDER BY name; + +CREATE TABLE IF NOT EXISTS _effect_orm_migration_steps ( + name String, step String, sql_hash String, finished_at DateTime64(3) DEFAULT now64(3) +) ENGINE = ReplacingMergeTree(finished_at) ORDER BY (name, step); +``` + +- A step is journaled only after it, and any mutation it started, has finished. The migration + row is written last. A rerun skips journaled steps. +- `hash` is sha256 of the rendered migration. With `strict`, an applied migration whose hash + differs is a `MigrateHashMismatch` error; otherwise it is a warning. +- Postgres: one transaction per migration, `pg_advisory_xact_lock` instead of `LOCK TABLE`, the + same ledger columns in a `effect_orm` schema, and an importer for an existing + `__drizzle_migrations` / `effect_sql_migrations` table so adopters keep their history. +- Errors are `Schema.TaggedError`s with namespaced tags (`MigrateLeaseHeld`, `MigrateHashMismatch`, + `MigrateStepFailed` with a `Schema.Defect` cause, `MigrateUnsupportedChange`), consistent with + how the package already reports `InvalidLiteral`. No throws. +- Each migration and step gets a span (`effect_orm.migration.name`, `effect_orm.migration.step`), + following the `Migrator ${id}_${name}` precedent. +- `Migrate.layer(options)` mirrors `ClickhouseMigrator.layer` for "migrate on startup". + +`Migrate.verify` introspects (`system.tables`, `system.columns`, `system.data_skipping_indices`, +`create_table_query` through `formatQuery()`) and diffs the live schema against the snapshot **of +the last applied migration**, not the code's HEAD. That reports drift and partial applies honestly. + +### 4.6 CLI + +One bin, `effect-orm` (next to `ch-bench`): + +| Command | Needs a server | Notes | +| ------------------------ | -------------- | ----------------------------------------------------------- | +| `generate` | no | 4.4 | +| `check` | no | DAG validity, recomputed ids, name/DAG order, commutativity, backfill convergence lint, shipped-migration immutability against a git base (`--base origin/main`) | +| `migrate` | yes | Runs `Migrate.run` with a `SqlClient` layer exported from the config file | +| `status` / `plan` | yes | Pending migrations and their rendered steps, no execution | +| `verify` | yes | 4.5 | +| `push` | yes | Dev only: introspect, diff, apply, no files | +| `pull [--init]` | yes | Introspect into `defineTable` source; `--init` baselines | + +The config file exports the `SqlClient` layer, so the package never imports a driver: + +```ts +export default defineConfig({ + dialect: "clickhouse", + schema: "./src/schema/*.ts", + out: "./migrations", + client: ClickhouseClient.layerConfig({ url: Config.String("CLICKHOUSE_URL") }), +}) +``` + +### 4.7 Packaging + +- Subpaths in the existing package: `./schema` (pure, browser-safe), `./migrate` (runtime, + Effect only), `./kit` (generator, diff, node `fs`), plus the bin. `./benchmark/cli` already sets + the precedent for node-only subpaths. +- No new runtime dependencies. Prompts use `node:readline`. +- The root barrel does not re-export `./kit` or `./migrate`, so query-only consumers pay nothing. +- `scripts/check-exports-documented.mjs` will demand docs for every new export. Keep the public + surface small: the doc pages are part of the work, not an afterthought. + +--- + +## 5. Phases + +0. **Validate assumptions.** Live-ClickHouse tests (the `test:release` matrix) for: Effect + `Migrator` behavior on a failing ClickHouse migration; `formatQuery()` round-trips of MV bodies + on the oldest supported server; `EXCHANGE TABLES` and `MODIFY QUERY` availability per version. +1. **Schema layer.** `./schema` with `defineTable` and `materializedView`, DDL rendering, engine + flavor and cluster options. Tests: rendered DDL is accepted by every matrix server, a + `defineTable` value compiles through every existing query path, a view's type-level check + rejects a mismatched target. +2. **Snapshot and offline generate (additive only).** Create table and view, add column, add + index, TTL, defaults. `check` with DAG ordering. Snapshot tests. +3. **Migrator.** Ledger, step journal, lease, hashes, `status`, `plan`, `migrate`, `verify`. + Live tests: kill between steps and resume, rerun is a no-op, hash mismatch fails under strict, + a replicated ledger under `ON CLUSTER` (matrix permitting). +4. **The hard ClickHouse changes.** Renames with hints, view recreation, type changes with + mutation waits, `rebuild`, windowed `backfill`, drops with dependency ordering, server-version + gated entities. +5. **`pull`, `push`, `--init`.** Introspection back into TypeScript; baseline adoption. +6. **Postgres.** Same envelope, transactional runtime, importers for drizzle and Effect ledgers. + Possibly only `migrate`/`verify` at first, if consumers keep drizzle-kit for authoring. +7. **First real consumer.** Maple's warehouse is the obvious one; it defines its schema with + another tool today, so adoption needs either an adapter or moving its definitions to + `defineTable`. That is Maple's decision and out of scope here, but its history is a useful + requirements check: chunked backfills, view-before-table drops, view bodies frozen per version, + and performance-only objects that must not block correctness. + +--- + +## 6. Open questions + +1. **Postgres scope.** Full parity with drizzle-kit, or ClickHouse-only authoring plus a runtime + that can also apply drizzle-kit folders? The second is much less work and avoids competing with + the tool people already use for Postgres. +2. **Packaging.** Subpaths of `@maple-dev/effect-orm` (proposed) or a separate + `effect-orm-kit` like drizzle-kit, which keeps the main package's install small. The repo has one + package today. +3. **`defineTable` vs extending `table()`.** A separate constructor keeps `table()` untouched; an + options argument on `table()` is less API. Proposed: separate, because a DDL table needs engine + and keys that a query-only table should not be forced to declare. +4. **Timestamps vs integers.** Proposed: Drizzle's timestamp names plus the `prevIds` DAG. Integer + ids are friendlier for "schema version N" gates in consumers; the runner can expose an ordinal + for that purpose either way. +5. **Canonical SQL offline.** `generate` cannot ask a server for `formatQuery()`. Is our printer + close enough that drift only matters in `verify`, or should `generate` optionally use a local + server or chDB? +6. **MV changes.** `DROP` + `CREATE` loses inserts during the gap; `MODIFY QUERY` has version and + setting constraints. Which is the default? +7. **Lease semantics** on ClickHouse without a coordinator: is best effort acceptable, or should + `ON CLUSTER` deployments require a Keeper-backed lock? +8. **Upstream.** Report the Effect `Migrator` ClickHouse behavior from phase 0 to Effect, and + whether `migrate` should be offered back as a ClickHouse-aware `ClickhouseMigrator`. + +--- + +## 7. Implementation notes (phases 0 to 3) + +**Phase 0 findings.** + +- Effect's `ClickhouseMigrator` (`@effect/sql-clickhouse` 4.0.0) fails on ClickHouse 26.8 before + running any migration: its generic ledger `CREATE TABLE` is a syntax error there. Worse than + section 2 assumed, and worth reporting upstream. +- `formatQuery` normalizes what `verify` needs (`INTERVAL 30 DAY` becomes `toIntervalDay(30)`), + and `defaultValueOfTypeName` normalizes types (`DateTime64` becomes `DateTime64(3)`). The + server rewrites codecs (`Delta` becomes `Delta(8)`), so codecs are not compared yet. +- `as_select` qualifies tables with the database name; `verify` strips it before comparing. +- `ALTER ... MODIFY TTL ... SETTINGS materialize_ttl_after_modify = 0`, `RESET SETTING`, + `REMOVE DEFAULT`, `REMOVE CODEC`, and `DROP COLUMN ... SETTINGS mutations_sync = 2` all work + on 26.2 and 26.8. + +**Departures from the plan.** + +- Generated migrations are `migration.json` (typed ops), not `migration.sql`. Ops render when + they run, so `ON CLUSTER` and `Replicated*` engines come from the deployment, not from the + committed file. Hand-written migrations (`--custom`) are `migration.sql` with drizzle-kit's + `--> statement-breakpoint`. The hash covers the canonical ops, so reformatting the JSON is not + an edit. +- The migrator takes a `MigrationDriver` (execute, query) instead of `SqlClient` directly, with + `fromSqlClient(sql, { command })` as the adapter: ClickHouse DDL has to go through the + client's `asCommand`, and the query path fails on statements with no result. +- `.ts` migrations that export an Effect are not implemented. +- `verify` checks tables, engine family, keys, columns (type, default), skipping indexes, and + view targets and bodies. TTL, codecs, settings, and comments are not compared yet. +- `check` treats two branches as conflicting when they change the same table (columns and + indexes count as their table), not per column as drizzle-kit's footprints do. Coarser, simpler, + and safe. +- `generate` bumps the timestamp past any prefix already in the folder, so two migrations never + share a second; `check` rejects a migration that sorts before its parent. + +**Not done (phase 4 onward).** Rename detection, column type changes, table rebuilds, windowed +backfills, `minServerVersion` entities, `pull` / `push` / `--init`, Postgres. `generate` +reports the unsupported changes by name and writes nothing, rather than guessing. + +**Tests.** `src/schema/schema.test.ts` and `src/kit/*.test.ts` run offline (definitions, +rendering, diff, snapshots, the CLI in a temp folder, branch analysis). +`tests/migrate.clickhouse.test.ts` runs against a live server: apply and verify clean, an +additive change with a recreated view checked with real inserts, resume after a failed +statement, hash mismatch under `strict`, the lease, and drift. Passing on 26.2.19.43 and 26.8.2.7. +`tests/package-consumer.mts` imports all three entry points from the packed tarball under Node. diff --git a/docs/README.md b/docs/README.md index 0585161..669926c 100644 --- a/docs/README.md +++ b/docs/README.md @@ -16,7 +16,8 @@ You do not need a Maple account, Maple's schema, or tenant columns. Tenant analy optional feature for applications that share tables between tenants. The root builder does not manage connections, create tables, run migrations, insert rows, or provide -an ORM. It does not validate SQL against a live server, choose query plans, enforce authorization, +an ORM. Opt-in [schema and migration entry points](./migrations.md) add DDL and migrations for +ClickHouse. It does not validate SQL against a live server, choose query plans, enforce authorization, or supply retries. Existing ClickHouse tables and your executor own those responsibilities. [Getting started](./getting-started.md) covers npm installation and building from source. @@ -53,6 +54,7 @@ Roughly in reading order. | [Tenant scoping](./tenant-scoping.md) | `tenantScope`, what marks a query scoped, `crossTenant()` | | [Extending the DSL](./extending.md) | `defineFn`, raw escape hatches, handwritten SQL | | [Postgres](./postgres.md) | The Postgres dialect, its column types and functions | +| [Schema and migrations](./migrations.md) | `defineTable`, `materializedView`, `effect-orm generate`, applying migrations | ## Reference @@ -72,6 +74,9 @@ Roughly in reading order. | `@maple-dev/effect-orm/benchmark` | Driver-free suite definitions, runner, report schemas, and comparisons | | `@maple-dev/effect-orm/benchmark/http` | ClickHouse HTTP transport, environment configuration, and query-log collection | | `@maple-dev/effect-orm/benchmark/cli` | `runCli(args)` for embedding the bundled `ch-bench` commands | +| `@maple-dev/effect-orm/schema` | `defineTable`, `materializedView`, DDL rendering, snapshots, and the schema diff. Pure | +| `@maple-dev/effect-orm/kit` | `generate` and `check` over a migrations folder, `defineConfig`, and `runCli` for the bundled `effect-orm` command. Node or Bun | +| `@maple-dev/effect-orm/migrate` | `run`, `status`, `verify`, and `MigrationDriver`: applies migrations through a driver you provide | The root barrel is curated, not exhaustive — see [the reference](./reference.md#whats-only-on-a-subpath) for what lives only on a subpath. diff --git a/docs/migrations.md b/docs/migrations.md new file mode 100644 index 0000000..283a6c4 --- /dev/null +++ b/docs/migrations.md @@ -0,0 +1,171 @@ +# Schema and migrations + +The query builder works with tables you manage elsewhere. If you would rather keep the schema +in TypeScript too, three entry points add that, all opt-in: + +| Entry | Runs where | What it does | +| ---------------------------------- | ------------------ | ----------------------------------------------------------------------------- | +| `@maple-dev/effect-orm/schema` | anywhere, pure | `defineTable` / `materializedView`, DDL rendering, snapshots, the diff | +| `@maple-dev/effect-orm/kit` | Node or Bun | `generate` and `check` over a migrations folder; the `effect-orm` command | +| `@maple-dev/effect-orm/migrate` | anywhere Effect runs | applies migrations through a driver you provide, `status`, `verify` | + +ClickHouse only, for now. The model follows drizzle-kit (a committed snapshot per migration, +an offline `generate`, data-loss confirmations by prompt or by hints), and the runtime is +built for a database without transactions. + +## Defining tables + +`defineTable` returns a `Table`, so every query API accepts it. Columns are the usual column +types, or `S.column(type, options)` for a default, a codec, or a comment. Keys, TTL, defaults, +and index expressions are SQL strings or DSL callbacks. + +```ts title="migrations-schema.ts" +import * as CH from "@maple-dev/effect-orm" +import * as S from "@maple-dev/effect-orm/schema" + +export const Requests = S.defineTable("requests", { + columns: { + OrgId: CH.custom("LowCardinality(String)", CH.string.schema), + Timestamp: CH.dateTime, + Route: CH.string, + Status: S.column(CH.uint16, { default: 200 }), + }, + engine: S.engine.mergeTree(), + orderBy: ["OrgId", "Route", "Timestamp"], + partitionBy: "toDate(Timestamp)", + ttl: S.ttlAfterDays("toDate(Timestamp)", 30), + indexes: [S.index("idx_status", ($) => $.Status, "set(100)")], + tenantColumn: "OrgId", +}) + +export const RoutesHourly = S.defineTable("routes_hourly", { + columns: { OrgId: CH.string, Hour: CH.dateTime, Route: CH.string, Requests: CH.uint64 }, + engine: S.engine.summingMergeTree(), + orderBy: ["OrgId", "Hour", "Route"], +}) + +// The body is a DSL query. An output column the target lacks, or of another +// type, is a type error here rather than a failed insert later. +export const RoutesHourlyMv = S.materializedView("routes_hourly_mv", { + to: RoutesHourly, + as: CH.from(Requests) + .select(($) => ({ OrgId: $.OrgId, Hour: CH.toStartOfHour($.Timestamp), Route: $.Route, Requests: CH.count() })) + .groupBy("OrgId", "Hour", "Route"), +}) + +export const ddl = S.renderSchema(S.entitiesOf([Requests, RoutesHourly, RoutesHourlyMv])) +``` + +A definition that cannot become DDL (a MergeTree without `orderBy`, a name that is not a plain +identifier, a view writing to a table outside the schema) throws `SchemaDefinitionDefect` when +the module loads. `CH.dateTime64` renders as `DateTime64`, which ClickHouse reads as +`DateTime64(3)`; declare another precision with `CH.custom`. + +Write engines as the plain family. Replicated engines and `ON CLUSTER` are render options +(`{ replicated: {}, cluster: "main" }`), so one schema serves a single server, a cluster, and +ClickHouse Cloud. + +## Generating migrations + +Create `effect-orm.config.ts`: + +```ts +import { defineConfig } from "@maple-dev/effect-orm/kit" + +export default defineConfig({ + schema: "./src/schema.ts", + out: "./migrations", +}) +``` + +Then `effect-orm generate --name add_status` (with Bun, or Node with type stripping) imports the +schema modules, diffs them against the newest snapshot in `out`, and writes +`migrations/_add_status/` holding `migration.json` and `snapshot.json`. It never +connects to a database. The plan it prints labels each statement: + +- `metadata`: a schema change with no data rewrite. +- `ingest gap`: a materialized view is dropped and recreated. Inserts in between are not + materialized by it. A view's body is fixed at creation, so this is the only way to change one. +- `destructive`: a table or column is dropped. + +Drops need confirmation. In a terminal, `generate` asks. Without one, it exits with status 2, +writes nothing, and prints the hints to pass back: + +```sh +effect-orm generate --hints '[{"type":"confirm_data_loss","kind":"column","entity":"requests.Route"}]' +``` + +Changes ClickHouse cannot make with `ALTER` (engine, sorting key, partition key, primary key, +column type) are reported and nothing is written. They need a table rebuild, which `generate` +does not write yet. Renames are not detected yet either: a rename reads as a drop plus an add, +and the drop asks for confirmation, so it never loses data silently. + +`effect-orm generate --custom` writes an empty `migration.sql` for statements you write by hand +(separate them with a line holding `--> statement-breakpoint`). Its snapshot copies its parent's. + +`effect-orm check` validates the folder: every snapshot id matches its contents, every parent +exists and sorts earlier, and branches merged from different pull requests touch different +tables. Independent branches are fine: the next `generate` records both as parents. Branches that +change the same table conflict; delete one migration and generate it again on top of the other. +Run `check` in CI. + +## Applying migrations + +The library opens no connection. Give it a `MigrationDriver`, usually built from the +`SqlClient` you query with. ClickHouse DDL has to go through the client's `asCommand`. + +```ts title="migrations-run.ts" +import { ClickhouseClient } from "@effect/sql-clickhouse" +import { Effect, Layer } from "effect" +import * as Migrate from "@maple-dev/effect-orm/migrate" + +const Driver = Layer.effect( + Migrate.MigrationDriver, + Effect.gen(function* () { + const sql = yield* ClickhouseClient.ClickhouseClient + return Migrate.fromSqlClient(sql, { command: sql.asCommand }) + }), +) + +export const program = Effect.gen(function* () { + const migrations = yield* Migrate.fromRecord({ + "20261003120000_init": { + kind: "sql", + migration: "CREATE TABLE IF NOT EXISTS t (x UInt8) ENGINE = MergeTree ORDER BY x", + }, + }) + const applied = yield* Migrate.run({ migrations, strict: true }) + const { drift } = yield* Migrate.verify(migrations) + return { applied, drift } +}).pipe( + Effect.provide(Driver), + Effect.provide(ClickhouseClient.layer({ url: "http://localhost:8123" })), +) +``` + +`Migrate.fromFileSystem(dir)` reads a migrations folder through Effect's `FileSystem`; add a +`driver` layer to the config and the CLI runs `effect-orm migrate`, `status`, and `verify`. + +How a run behaves: + +- **Order** follows the snapshots' parent links, then names. +- **Pending** means not in the ledger, by name. An applied migration's hash is checked; with + `strict` a changed file fails with `MigrateHashMismatch`, otherwise it logs a warning. +- **Each statement is journaled** in `_effect_orm_migration_steps` after it finishes, and the + migration row in `_effect_orm_migrations` is written last. A run that fails partway leaves + the migration `partial`; the next run skips the finished statements and resumes. +- **A lease** in `_effect_orm_migration_lease` stops a second run while one is active. It is best + effort: two runs starting in the same instant can both proceed. Serialize deploys if that + matters. +- **No rollback.** Write a new migration. Prefer expand, then contract: add the new shape, move + readers, and drop the old shape in a later migration. + +`Migrate.verify` compares the database with the snapshot of the **last applied** migration, so a +database that is behind is reported as behind (`status`), not as drifted. The server normalizes +both sides (`formatQuery`, `defaultValueOfTypeName`). It checks tables, engine family, keys, +columns, skipping indexes, and view targets and bodies; it does not check TTL, codecs, settings, +or comments yet. `effect-orm verify` exits 3 when it finds drift. + +_(Effect's own `ClickhouseMigrator` creates its ledger with a statement ClickHouse 26.8 +rejects, and inserts ledger rows before running each migration. That is why this package has +its own runner.)_ diff --git a/package.json b/package.json index 33b27e2..057bede 100644 --- a/package.json +++ b/package.json @@ -22,7 +22,8 @@ "url": "git+https://github.com/MapleTechLabs/effect-orm.git" }, "bin": { - "ch-bench": "./dist/benchmark/bin.mjs" + "ch-bench": "./dist/benchmark/bin.mjs", + "effect-orm": "./dist/kit/bin.mjs" }, "files": [ "dist", @@ -52,6 +53,18 @@ "types": "./dist/postgres.d.mts", "import": "./dist/postgres.mjs" }, + "./schema": { + "types": "./dist/schema.d.mts", + "import": "./dist/schema.mjs" + }, + "./migrate": { + "types": "./dist/migrate.d.mts", + "import": "./dist/migrate.mjs" + }, + "./kit": { + "types": "./dist/kit.d.mts", + "import": "./dist/kit.mjs" + }, "./benchmark": { "types": "./dist/benchmark/index.d.mts", "import": "./dist/benchmark/index.mjs" diff --git a/scripts/check-package.ts b/scripts/check-package.ts index a545df7..8c100a8 100644 --- a/scripts/check-package.ts +++ b/scripts/check-package.ts @@ -86,6 +86,7 @@ const program = Effect.gen(function* () { temporary, true, ) + yield* runCommand(path.join(temporary, "node_modules/.bin/effect-orm"), ["--help"], temporary, true) }) // This is the CLI entry point; provide platform services once at the boundary. diff --git a/src/kit.ts b/src/kit.ts new file mode 100644 index 0000000..d3b23cd --- /dev/null +++ b/src/kit.ts @@ -0,0 +1,19 @@ +// @maple-dev/effect-orm/kit +// +// The authoring side, for Node or Bun: read the schema modules and the +// migrations folder, write the next migration, check the folder. The +// `effect-orm` bin runs these. See docs/migrations.md. + +export { analyze, type GraphAnalysis, type GraphProblem } from "./kit/graph" +export { + KitError, + check, + defineConfig, + generate, + loadSchema, + readMigrations, + type GenerateOptions, + type GenerateResult, + type KitConfig, +} from "./kit/generate" +export { runCli } from "./kit/cli" diff --git a/src/kit/bin.ts b/src/kit/bin.ts new file mode 100644 index 0000000..b84b957 --- /dev/null +++ b/src/kit/bin.ts @@ -0,0 +1,3 @@ +#!/usr/bin/env node +import { runCli } from "./cli" +process.exitCode = await runCli(process.argv.slice(2)) diff --git a/src/kit/cli.ts b/src/kit/cli.ts new file mode 100644 index 0000000..91657f0 --- /dev/null +++ b/src/kit/cli.ts @@ -0,0 +1,158 @@ +import { resolve } from "node:path" +import { createInterface } from "node:readline/promises" +import { pathToFileURL } from "node:url" +import { parseArgs } from "node:util" +import { Effect, Schema } from "effect" +import { readFile } from "node:fs/promises" +import * as Migrate from "../migrate" +import { Hints, type Hint } from "../schema/diff" +import { check, generate, KitError, readMigrations, type KitConfig } from "./generate" + +const help = `effect-orm: schema migrations for ClickHouse + + generate [--name x] [--custom] Diff the schema against the migrations folder and write the next migration + [--hints ] [--hints-file ] + check Validate the migrations folder (snapshot chain, branch conflicts) + migrate [--strict] Apply pending migrations (needs config.driver) + status Applied, pending, partial, or changed, per migration + verify Compare the database with the last applied snapshot + + --config Default: effect-orm.config.ts + --json Machine-readable output + +Exit codes: 0 ok, 1 error, 2 missing hints (confirm data loss), 3 drift found.` + +const options = { + config: { type: "string" }, + name: { type: "string" }, + custom: { type: "boolean" }, + hints: { type: "string" }, + "hints-file": { type: "string" }, + strict: { type: "boolean" }, + json: { type: "boolean" }, + help: { type: "boolean" }, +} as const + +const decodeHints = Schema.decodeUnknownEffect(Schema.fromJsonString(Hints)) + +const loadConfig = (path: string) => + Effect.tryPromise({ + try: () => import(pathToFileURL(path).href) as Promise<{ default?: KitConfig }>, + catch: (cause) => new KitError({ code: "config", message: `Cannot import ${path}: ${cause instanceof Error ? cause.message : String(cause)}` }), + }).pipe( + Effect.flatMap((module) => + module.default === undefined + ? Effect.fail(new KitError({ code: "config", message: `${path} has no default export; export defineConfig({ ... })` })) + : Effect.succeed(module.default), + ), + ) + +const promptConfirm = (hint: Hint): Effect.Effect => + Effect.promise(async () => { + const rl = createInterface({ input: process.stdin, output: process.stdout }) + const answer = await rl.question(`Drop ${hint.kind} ${hint.entity}? Its data will be lost. [y/N] `) + rl.close() + return /^y(es)?$/i.test(answer.trim()) + }) + +const withDriver = (config: KitConfig, body: Effect.Effect) => + config.driver === undefined + ? Effect.fail(new KitError({ code: "config", message: "this command needs a database: set `driver` in the config" })) + : body.pipe(Effect.provide(config.driver)) + +const program = (args: ReadonlyArray, print: (line: string) => void) => + Effect.gen(function* () { + const parsed = yield* Effect.try({ + try: () => parseArgs({ args: [...args], options, allowPositionals: true }), + catch: (cause) => new KitError({ code: "config", message: cause instanceof Error ? cause.message : String(cause) }), + }) + const [command] = parsed.positionals + if (parsed.values.help === true || command === undefined) { + print(help) + return 0 + } + const cwd = process.cwd() + const config = yield* loadConfig(resolve(cwd, parsed.values.config ?? "effect-orm.config.ts")) + const json = parsed.values.json === true + const out = (value: unknown, text: () => ReadonlyArray) => { + if (json) print(JSON.stringify(value)) + else for (const line of text()) print(line) + } + + switch (command) { + case "generate": { + const fromFile = parsed.values["hints-file"] + const rawHints = + parsed.values.hints ?? + (fromFile === undefined ? undefined : yield* Effect.tryPromise({ try: () => readFile(resolve(cwd, fromFile), "utf8"), catch: () => new KitError({ code: "io", message: `Cannot read ${fromFile}` }) })) + const hints = rawHints === undefined ? [] : yield* decodeHints(rawHints).pipe(Effect.mapError(() => new KitError({ code: "config", message: "--hints must be a JSON array of hints" }))) + const interactive = !json && process.stdin.isTTY === true + const result = yield* generate(config, cwd, { + hints, + ...(parsed.values.name !== undefined ? { name: parsed.values.name } : undefined), + ...(parsed.values.custom === true ? { custom: true } : undefined), + ...(interactive ? { confirm: promptConfirm } : undefined), + }) + out(result, () => + result.written === undefined + ? ["No schema changes, nothing to write."] + : [`Wrote ${result.written}`, ...result.plan.flatMap((step) => step.sql.map((sql) => ` [${step.label}] ${sql.replace(/\n/g, "\n ")}`))], + ) + return 0 + } + case "check": { + const result = yield* check(config, cwd) + out(result, () => [ + `${result.migrations} migrations, ok.`, + ...(result.leaves.length > 1 ? [`Independent branches (merged by the next generate): ${result.leaves.join(", ")}`] : []), + ]) + return 0 + } + case "migrate": + case "status": + case "verify": { + const migrations = yield* readMigrations(resolve(cwd, config.out)) + const render = config.render ?? {} + if (command === "migrate") { + const ran = yield* withDriver(config, Migrate.run({ migrations, render, strict: parsed.values.strict === true })) + out(ran, () => (ran.length === 0 ? ["Nothing to apply."] : ran.map((m) => `applied ${m.name} (${m.steps} statements${m.resumedSteps > 0 ? `, ${m.resumedSteps} resumed` : ""})`))) + return 0 + } + if (command === "status") { + const rows = yield* withDriver(config, Migrate.status(migrations, render)) + out(rows, () => rows.map((r) => `${r.state.padEnd(8)} ${r.name}${r.appliedAt !== undefined ? ` ${r.appliedAt}` : ""}`)) + return 0 + } + const result = yield* withDriver(config, Migrate.verify(migrations)) + out(result, () => + result.against === undefined + ? ["No applied migration with a snapshot to compare against."] + : result.drift.length === 0 + ? [`No drift against ${result.against}.`] + : [`Drift against ${result.against}:`, ...result.drift.map((d) => ` ${d.entity}: ${d.problem}${d.expected !== undefined || d.actual !== undefined ? ` (expected ${d.expected ?? "-"}, actual ${d.actual ?? "-"})` : ""}`)], + ) + return result.drift.length === 0 ? 0 : 3 + } + default: + print(help) + return 1 + } + }) + +/** Runs the CLI and returns its exit code. */ +export const runCli = (args: ReadonlyArray, print: (line: string) => void = (line) => console.log(line)): Promise => + Effect.runPromise( + program(args, print).pipe( + Effect.catch((error: unknown) => + Effect.sync(() => { + if (error instanceof KitError) { + console.error(`effect-orm: ${error.message}`) + for (const line of error.details ?? []) console.error(` ${line}`) + return error.code === "missing_hints" ? 2 : 1 + } + console.error(`effect-orm: ${error instanceof Error ? error.message : String(error)}`) + return 1 + }), + ), + ), + ) diff --git a/src/kit/generate.ts b/src/kit/generate.ts new file mode 100644 index 0000000..1689adf --- /dev/null +++ b/src/kit/generate.ts @@ -0,0 +1,189 @@ +// `effect-orm generate`: diff the schema modules against the migrations folder +// and write the next migration. Offline: it never connects to a database. + +import { mkdir, readdir, readFile, rename, stat, writeFile } from "node:fs/promises" +import { join, resolve } from "node:path" +import { pathToFileURL } from "node:url" +import { Effect, Schema } from "effect" +import type { Layer } from "effect" +import { fromRecord, type LoadedMigration, type MigrationInput } from "../migrate/source" +import type { MigrationDriver } from "../migrate/driver" +import { diffSchemas, type Hint } from "../schema/diff" +import { labelOf, renderOp, type MigrationFile } from "../schema/ops" +import type { RenderOptions } from "../schema/render" +import { entitiesOf, isSchemaObject, makeSnapshot, serializeSnapshot, type SchemaObject } from "../schema/snapshot" +import { analyze, type GraphProblem } from "./graph" + +export interface KitConfig { + /** Modules whose exports include `defineTable` / `materializedView` values. */ + readonly schema: string | ReadonlyArray + /** The migrations folder. */ + readonly out: string + /** How `migrate`, `status`, and `verify` render statements for this deployment. */ + readonly render?: RenderOptions + /** Needed by `migrate`, `status`, and `verify`. Build it from your `SqlClient`. */ + readonly driver?: Layer.Layer +} + +/** Typed identity, for `effect-orm.config.ts`. */ +export const defineConfig = (config: KitConfig): KitConfig => config + +/** Anything the kit could not do. `code` decides the CLI's exit status. */ +export class KitError extends Schema.TaggedError()("@maple-dev/effect-orm/KitError", { + code: Schema.Literals(["config", "io", "check", "unsupported", "missing_hints"]), + message: Schema.String, + details: Schema.optional(Schema.Array(Schema.String)), +}) {} + +const io = (op: () => Promise, message: string) => + Effect.tryPromise({ try: op, catch: (cause) => new KitError({ code: "io", message: `${message}: ${cause instanceof Error ? cause.message : String(cause)}` }) }) + +/** Import every schema module and collect the definitions they export. */ +export const loadSchema = (config: KitConfig, cwd: string): Effect.Effect, KitError> => + Effect.gen(function* () { + const paths = typeof config.schema === "string" ? [config.schema] : config.schema + const objects = new Set() + for (const path of paths) { + const module = yield* io( + () => import(pathToFileURL(resolve(cwd, path)).href) as Promise>, + `Cannot import ${path} (TypeScript needs Bun, or Node with type stripping)`, + ) + for (const value of Object.values(module)) if (isSchemaObject(value)) objects.add(value) + } + return [...objects] + }) + +/** Read `//...` with node:fs. */ +export const readMigrations = (out: string): Effect.Effect, KitError> => + Effect.gen(function* () { + const exists = yield* io(() => stat(out).then(() => true, () => false), `Cannot stat ${out}`) + if (!exists) return [] + const names = yield* io(() => readdir(out), `Cannot read ${out}`) + const record: Record = {} + const optional = (path: string) => readFile(path, "utf8").then((text): string | undefined => text, () => undefined) + for (const name of names.sort()) { + const dir = join(out, name) + const isDir = yield* io(() => stat(dir).then((s) => s.isDirectory()), `Cannot stat ${dir}`) + if (!isDir) continue + const json = yield* io(() => optional(join(dir, "migration.json")), `Cannot read ${dir}`) + const sql = yield* io(() => optional(join(dir, "migration.sql")), `Cannot read ${dir}`) + const snapshot = yield* io(() => optional(join(dir, "snapshot.json")), `Cannot read ${dir}`) + const migration = json ?? sql + if (migration === undefined) continue + record[name] = { migration, kind: json !== undefined ? "ops" : "sql", ...(snapshot !== undefined ? { snapshot } : undefined) } + } + return yield* fromRecord(record).pipe(Effect.mapError((e) => new KitError({ code: "check", message: `${e.migration}: ${e.message}` }))) + }) + +const describeProblems = (problems: ReadonlyArray) => problems.map((p) => `${p.migration}: ${p.message}`) + +/** `effect-orm check`. */ +export const check = (config: KitConfig, cwd: string) => + Effect.gen(function* () { + const migrations = yield* readMigrations(resolve(cwd, config.out)) + const analysis = yield* analyze(migrations) + if (analysis.problems.length > 0) { + return yield* new KitError({ code: "check", message: "the migrations folder is inconsistent", details: describeProblems(analysis.problems) }) + } + return { migrations: migrations.length, leaves: analysis.leaves.map((l) => l.name) } + }) + +const ADJECTIVES = ["amber", "brisk", "calm", "deft", "eager", "fond", "gentle", "hardy", "keen", "lucid", "merry", "nimble", "quiet", "rapid", "steady", "tidy", "vivid", "witty"] +const NOUNS = ["badger", "comet", "delta", "ember", "falcon", "glacier", "harbor", "island", "juniper", "kestrel", "lantern", "meadow", "nebula", "otter", "prairie", "quartz", "river", "summit"] + +const pick = (list: ReadonlyArray): A => list[Math.floor(Math.random() * list.length)]! + +/** `YYYYMMDDHHMMSS` in UTC, bumped past any prefix already in use so two migrations never share a second. */ +const timestamp = (now: Date, taken: ReadonlySet): string => { + let t = Math.floor(now.getTime() / 1000) * 1000 + for (;;) { + const d = new Date(t) + const p = (n: number, w = 2) => String(n).padStart(w, "0") + const prefix = `${p(d.getUTCFullYear(), 4)}${p(d.getUTCMonth() + 1)}${p(d.getUTCDate())}${p(d.getUTCHours())}${p(d.getUTCMinutes())}${p(d.getUTCSeconds())}` + if (!taken.has(prefix)) return prefix + t += 1000 + } +} + +export interface GenerateOptions { + readonly name?: string + readonly custom?: boolean + readonly hints?: ReadonlyArray + /** Asked for each data-loss confirmation the hints do not cover. Absent: report them as missing. */ + readonly confirm?: (hint: Hint) => Effect.Effect + readonly now?: Date +} + +export interface GenerateResult { + readonly written: string | undefined + readonly plan: ReadonlyArray<{ readonly label: string; readonly sql: ReadonlyArray }> +} + +/** `effect-orm generate`. */ +export const generate = (config: KitConfig, cwd: string, options: GenerateOptions = {}) => + Effect.gen(function* () { + const out = resolve(cwd, config.out) + const migrations = yield* readMigrations(out) + const analysis = yield* analyze(migrations) + if (analysis.problems.length > 0) { + return yield* new KitError({ code: "check", message: "fix the migrations folder first", details: describeProblems(analysis.problems) }) + } + if (options.name !== undefined && !/^[a-z0-9_]+$/.test(options.name)) { + return yield* new KitError({ code: "config", message: "--name takes lowercase letters, digits, and underscores" }) + } + const objects = yield* loadSchema(config, cwd) + const next = yield* Effect.try({ + try: () => entitiesOf(objects), + catch: (cause) => new KitError({ code: "config", message: cause instanceof Error ? cause.message : String(cause) }), + }) + + let file: MigrationFile | undefined + let snapshotEntities = next + if (options.custom === true) { + snapshotEntities = analysis.base + } else { + const hints = [...(options.hints ?? [])] + let diff = diffSchemas(analysis.base, next, hints) + if (diff.unsupported.length > 0) { + return yield* new KitError({ + code: "unsupported", + message: "these changes need a table rebuild or a data rewrite, which generate does not write yet", + details: diff.unsupported.map((u) => `${u.entity}: ${u.message}`), + }) + } + if (diff.missingHints.length > 0 && options.confirm !== undefined) { + for (const hint of diff.missingHints) { + if (yield* options.confirm(hint)) hints.push(hint) + } + diff = diffSchemas(analysis.base, next, hints) + } + if (diff.missingHints.length > 0) { + return yield* new KitError({ + code: "missing_hints", + message: "confirm data loss with --hints, or run in a terminal to be asked", + details: [JSON.stringify(diff.missingHints)], + }) + } + if (diff.ops.length === 0) return { written: undefined, plan: [] } satisfies GenerateResult + file = { version: "1", ops: diff.ops } + } + + const snapshot = yield* makeSnapshot(snapshotEntities, analysis.baseIds) + const taken = new Set(migrations.map((m) => m.name.slice(0, 14))) + const name = `${timestamp(options.now ?? new Date(), taken)}_${options.name ?? `${pick(ADJECTIVES)}_${pick(NOUNS)}`}` + const dir = join(out, name) + const staging = `${dir}.tmp-${process.pid}` + yield* io(() => mkdir(staging, { recursive: true }), `Cannot create ${staging}`) + yield* io( + () => + file === undefined + ? writeFile(join(staging, "migration.sql"), "-- Custom SQL migration. Separate statements with a line holding only:\n-- --> statement-breakpoint\n") + : writeFile(join(staging, "migration.json"), `${JSON.stringify(file, null, "\t")}\n`), + `Cannot write ${dir}`, + ) + yield* io(() => writeFile(join(staging, "snapshot.json"), serializeSnapshot(snapshot)), `Cannot write ${dir}`) + yield* io(() => rename(staging, dir), `Cannot move ${staging} to ${dir}`) + + const plan = (file?.ops ?? []).map((op) => ({ label: labelOf(op), sql: renderOp(op, config.render) })) + return { written: dir, plan } satisfies GenerateResult + }) diff --git a/src/kit/graph.test.ts b/src/kit/graph.test.ts new file mode 100644 index 0000000..c6e5f82 --- /dev/null +++ b/src/kit/graph.test.ts @@ -0,0 +1,65 @@ +import { Effect } from "effect" +import { describe, expect, it } from "vitest" +import * as CH from "../ch/index" +import { fromRecord, type MigrationInput } from "../migrate/source" +import * as S from "../schema" +import { analyze } from "./graph" + +const table = (name: string, extra: Record = {}) => + S.defineTable(name, { columns: { Id: CH.string, ...extra }, engine: S.engine.mergeTree(), orderBy: ["Id"] }) + +const input = (objects: ReadonlyArray, prevIds: ReadonlyArray) => + Effect.map(S.makeSnapshot(S.entitiesOf(objects), prevIds), (snapshot) => ({ + snapshot, + input: { kind: "ops", migration: '{"version":"1","ops":[]}', snapshot: S.serializeSnapshot(snapshot) } satisfies MigrationInput, + })) + +const run = (effect: Effect.Effect) => Effect.runPromise(effect) + +describe("analyze", () => { + it("merges branches that touch different tables", async () => { + const result = await run( + Effect.gen(function* () { + const root = yield* input([table("a")], [S.ORIGIN_ID]) + const left = yield* input([table("a"), table("b")], [root.snapshot.id]) + const right = yield* input([table("a"), table("c")], [root.snapshot.id]) + return yield* analyze( + yield* fromRecord({ "20260101000000_root": root.input, "20260102000000_left": left.input, "20260102000001_right": right.input }), + ) + }), + ) + expect(result.problems).toEqual([]) + expect(result.leaves.map((l) => l.name)).toEqual(["20260102000000_left", "20260102000001_right"]) + expect(result.baseIds).toHaveLength(2) + expect(result.base.filter((e) => e.kind === "table").map((e) => e.name).sort()).toEqual(["a", "b", "c"]) + }) + + it("reports branches that change the same table", async () => { + const result = await run( + Effect.gen(function* () { + const root = yield* input([table("a")], [S.ORIGIN_ID]) + const left = yield* input([table("a", { X: CH.string })], [root.snapshot.id]) + const right = yield* input([table("a", { Y: CH.string })], [root.snapshot.id]) + return yield* analyze( + yield* fromRecord({ "20260101000000_root": root.input, "20260102000000_left": left.input, "20260102000001_right": right.input }), + ) + }), + ) + expect(result.problems).toEqual([ + expect.objectContaining({ migration: "20260102000001_right", message: expect.stringContaining("both change table a") }), + ]) + }) + + it("reports a migration that sorts before its parent", async () => { + const result = await run( + Effect.gen(function* () { + const root = yield* input([table("a")], [S.ORIGIN_ID]) + const child = yield* input([table("a"), table("b")], [root.snapshot.id]) + return yield* analyze(yield* fromRecord({ "20260102000000_root": root.input, "20260101000000_child": child.input })) + }), + ) + expect(result.problems).toEqual([ + expect.objectContaining({ migration: "20260101000000_child", message: expect.stringContaining("sorts before its parent") }), + ]) + }) +}) diff --git a/src/kit/graph.ts b/src/kit/graph.ts new file mode 100644 index 0000000..16f7a5f --- /dev/null +++ b/src/kit/graph.ts @@ -0,0 +1,129 @@ +// The migration graph: what `check` validates and what `generate` diffs against. +// +// Mirrors drizzle-kit's `check`. Branches that touch different objects are +// commutative: the database ends up the same whichever lands first, so the +// next `generate` diffs against both merged and records both leaves as +// parents. Branches that touch the same object conflict, and one of them has +// to be regenerated on top of the other. + +import { Effect } from "effect" +import { canonicalJson, entityKey, ORIGIN_ID, sha256Hex, sortEntities, type SchemaEntity } from "../schema/entities" +import { migrationParents, type LoadedMigration } from "../migrate/source" + +export interface GraphProblem { + readonly migration: string + readonly message: string +} + +export interface GraphAnalysis { + readonly problems: ReadonlyArray + /** Migrations nothing builds on yet. More than one means unmerged branches. */ + readonly leaves: ReadonlyArray + /** The schema the next migration starts from. */ + readonly base: ReadonlyArray + /** Parents for the next migration's snapshot. */ + readonly baseIds: ReadonlyArray +} + +const ancestorsOf = ( + m: LoadedMigration, + parents: ReadonlyMap>, +): Set => { + const seen = new Set() + const stack = [m] + while (stack.length > 0) { + const next = stack.pop()! + if (seen.has(next)) continue + seen.add(next) + stack.push(...(parents.get(next) ?? [])) + } + return seen +} + +/** Entity changes from `from` to `to`: key -> new entity, or null for removed. */ +const changesBetween = ( + from: ReadonlyArray, + to: ReadonlyArray, +): Map => { + const before = new Map(from.map((e) => [entityKey(e), canonicalJson(e)])) + const after = new Map(to.map((e) => [entityKey(e), e])) + const out = new Map() + for (const [key, entity] of after) if (before.get(key) !== canonicalJson(entity)) out.set(key, entity) + for (const key of before.keys()) if (!after.has(key)) out.set(key, null) + return out +} + +/** The table an entity belongs to, so two branches editing one table conflict. */ +const ownerOf = (key: string): string => { + const [kind, rest = ""] = key.split(":") + return kind === "column" || kind === "index" ? `table:${rest.split(".")[0]}` : key +} + +export const analyze = (migrations: ReadonlyArray): Effect.Effect => + Effect.gen(function* () { + const problems: Array = [] + const withSnapshots = migrations.filter((m) => m.snapshot !== undefined) + for (const m of migrations) { + if (m.snapshot === undefined) problems.push({ migration: m.name, message: "has no snapshot.json" }) + } + const knownIds = new Set(withSnapshots.map((m) => m.snapshot!.id)) + for (const m of withSnapshots) { + const id = yield* Effect.promise(() => sha256Hex(canonicalJson(sortEntities(m.snapshot!.entities)))) + if (id !== m.snapshot!.id) { + problems.push({ migration: m.name, message: "snapshot id does not match its entities; the snapshot was edited by hand" }) + } + } + const parents = migrationParents(withSnapshots) + for (const m of withSnapshots) { + for (const id of m.snapshot!.prevIds) { + if (id === ORIGIN_ID) continue + if (!knownIds.has(id)) problems.push({ migration: m.name, message: `parent snapshot ${id.slice(0, 12)} is not in the folder` }) + else if (!(parents.get(m) ?? []).some((p) => p.snapshot!.id === id)) { + problems.push({ + migration: m.name, + message: `sorts before its parent ${id.slice(0, 12)}; rename the folder so its timestamp is later`, + }) + } + } + } + + // A broken chain makes leaves and ancestors meaningless; report it alone. + if (problems.length > 0) return { problems, leaves: [], base: [], baseIds: [] } + + const hasChild = new Set() + for (const list of parents.values()) for (const p of list) hasChild.add(p) + const leaves = withSnapshots.filter((m) => !hasChild.has(m)) + + if (leaves.length === 0) return { problems, leaves, base: [], baseIds: [ORIGIN_ID] } + if (leaves.length === 1) { + const leaf = leaves[0]! + return { problems, leaves, base: leaf.snapshot!.entities, baseIds: [leaf.snapshot!.id] } + } + + // Several leaves: merge them on their nearest common ancestor. + const ancestorSets = leaves.map((leaf) => ancestorsOf(leaf, parents)) + const common = withSnapshots.filter((m) => ancestorSets.every((set) => set.has(m))) + const ancestor = common.reduce( + (best, m) => (best === undefined || ancestorsOf(m, parents).size > ancestorsOf(best, parents).size ? m : best), + undefined, + ) + const ancestorEntities = ancestor?.snapshot!.entities ?? [] + const merged = new Map(ancestorEntities.map((e) => [entityKey(e), e] as const)) + const touchedBy = new Map() + for (const leaf of leaves) { + for (const [key, entity] of changesBetween(ancestorEntities, leaf.snapshot!.entities)) { + const owner = ownerOf(key) + const other = touchedBy.get(owner) + if (other !== undefined && other !== leaf.name) { + problems.push({ + migration: leaf.name, + message: `conflicts with ${other}: both change ${owner.replace(":", " ")}. Delete one and generate it again on top of the other`, + }) + } + touchedBy.set(owner, leaf.name) + if (entity === null) merged.delete(key) + else merged.set(key, entity) + } + } + return { problems, leaves, base: [...merged.values()], baseIds: leaves.map((l) => l.snapshot!.id) } + }) diff --git a/src/kit/kit.test.ts b/src/kit/kit.test.ts new file mode 100644 index 0000000..8df7ff9 --- /dev/null +++ b/src/kit/kit.test.ts @@ -0,0 +1,91 @@ +import { mkdtempSync, readdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs" +import { tmpdir } from "node:os" +import { join, resolve } from "node:path" +import { afterEach, beforeEach, describe, expect, it } from "vitest" +import { runCli } from "../kit" + +const src = resolve(import.meta.dirname, "..") + +const schemaModule = (extra: { column?: boolean; dropName?: boolean } = {}) => ` +import * as CH from "${src}/ch/index" +import * as S from "${src}/schema" + +export const Events = S.defineTable("events", { + columns: { + OrgId: CH.string, + Timestamp: CH.dateTime64, + ${extra.dropName === true ? "" : "Name: CH.string,"} + ${extra.column === true ? 'Env: S.column(CH.string, { default: "" }),' : ""} + }, + engine: S.engine.mergeTree(), + orderBy: ["OrgId", "Timestamp"], +}) +` + +describe("effect-orm CLI", () => { + let dir = "" + let lines: Array = [] + const cli = (...args: Array) => { + const cwd = process.cwd() + process.chdir(dir) + return runCli(args, (line) => lines.push(line)).finally(() => process.chdir(cwd)) + } + const folders = () => readdirSync(join(dir, "migrations")).sort() + + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), "effect-orm-kit-")) + lines = [] + writeFileSync(join(dir, "effect-orm.config.ts"), `export default { schema: "./schema.ts", out: "./migrations" }\n`) + }) + afterEach(() => rmSync(dir, { recursive: true, force: true })) + + it("generates a first migration, then nothing for an unchanged schema", async () => { + writeFileSync(join(dir, "schema.ts"), schemaModule()) + expect(await cli("generate", "--name", "init")).toBe(0) + const [first] = folders() + expect(first).toMatch(/^\d{14}_init$/) + const file = JSON.parse(readFileSync(join(dir, "migrations", first!, "migration.json"), "utf8")) + expect(file.ops.map((op: { op: string }) => op.op)).toEqual(["create_table"]) + + expect(await cli("generate")).toBe(0) + expect(lines.at(-1)).toBe("No schema changes, nothing to write.") + expect(folders()).toHaveLength(1) + expect(await cli("check")).toBe(0) + }) + + it("chains snapshots and never reuses a timestamp", async () => { + writeFileSync(join(dir, "schema.ts"), schemaModule()) + await cli("generate", "--name", "init") + writeFileSync(join(dir, "schema.ts"), `${schemaModule({ column: true })}\n// v2\n`) + // A fresh module path, so the import cache does not hand back v1. + renameSync(join(dir, "schema.ts"), join(dir, "schema2.ts")) + writeFileSync(join(dir, "v2.config.ts"), `export default { schema: "./schema2.ts", out: "./migrations" }\n`) + expect(await cli("generate", "--name", "env", "--config", "v2.config.ts")).toBe(0) + const [a, b] = folders() + expect(a!.slice(0, 14)).not.toBe(b!.slice(0, 14)) + const s1 = JSON.parse(readFileSync(join(dir, "migrations", a!, "snapshot.json"), "utf8")) + const s2 = JSON.parse(readFileSync(join(dir, "migrations", b!, "snapshot.json"), "utf8")) + expect(s2.prevIds).toEqual([s1.id]) + expect(lines.join("\n")).toContain("ADD COLUMN IF NOT EXISTS Env String DEFAULT '' AFTER Name") + }) + + it("exits 2 on data loss until it is confirmed with hints", async () => { + writeFileSync(join(dir, "schema.ts"), schemaModule()) + await cli("generate", "--name", "init") + writeFileSync(join(dir, "schema3.ts"), schemaModule({ dropName: true })) + writeFileSync(join(dir, "v3.config.ts"), `export default { schema: "./schema3.ts", out: "./migrations" }\n`) + expect(await cli("generate", "--json", "--config", "v3.config.ts")).toBe(2) + expect(folders()).toHaveLength(1) + const hints = JSON.stringify([{ type: "confirm_data_loss", kind: "column", entity: "events.Name" }]) + expect(await cli("generate", "--json", "--config", "v3.config.ts", "--hints", hints)).toBe(0) + expect(folders()).toHaveLength(2) + }) + + it("check fails when a snapshot was edited by hand", async () => { + writeFileSync(join(dir, "schema.ts"), schemaModule()) + await cli("generate", "--name", "init") + const path = join(dir, "migrations", folders()[0]!, "snapshot.json") + writeFileSync(path, readFileSync(path, "utf8").replace('"String"', '"UInt8"')) + expect(await cli("check")).toBe(1) + }) +}) diff --git a/src/migrate.ts b/src/migrate.ts new file mode 100644 index 0000000..905d9ae --- /dev/null +++ b/src/migrate.ts @@ -0,0 +1,29 @@ +// @maple-dev/effect-orm/migrate +// +// Applies migrations written by `effect-orm generate` (or by hand) to a +// ClickHouse database through a `MigrationDriver` you provide. Statements are +// journaled one by one, so a failed run resumes where it stopped. See +// docs/migrations.md. + +export { MigrationDriver, fromSqlClient, layerSqlClient, type FromSqlClientOptions, type MigrationDriverApi } from "./migrate/driver" +export { + MigrateHashMismatch, + MigrateLeaseHeld, + MigrateSourceError, + MigrateSqlError, + MigrateStepFailed, + type MigrateError, +} from "./migrate/errors" +export { LEDGER_TABLES } from "./migrate/ledger" +export { + STATEMENT_BREAKPOINT, + fromFileSystem, + fromRecord, + orderMigrations, + stepsOf, + type LoadedMigration, + type MigrationInput, + type MigrationStep, +} from "./migrate/source" +export { run, status, type AppliedMigration, type MigrationState, type MigrationStatus, type RunOptions } from "./migrate/run" +export { verify, type Drift, type VerifyResult } from "./migrate/verify" diff --git a/src/migrate/driver.ts b/src/migrate/driver.ts new file mode 100644 index 0000000..1a2e929 --- /dev/null +++ b/src/migrate/driver.ts @@ -0,0 +1,67 @@ +// The migrator's only dependency: run a statement, read rows. +// +// The library never opens a connection. A caller provides a driver, most often +// built from the `SqlClient` they already use for queries. ClickHouse DDL must +// go through the client's command path (`asCommand` on +// `@effect/sql-clickhouse`): its query path asks for a JSON result, which a +// DDL statement does not have. + +import { Context, Effect, Layer } from "effect" +import * as SqlClient from "effect/sql/SqlClient" +import { MigrateSqlError } from "./errors" + +export interface MigrationDriverApi { + /** Run a statement that returns no rows (DDL, INSERT). */ + readonly execute: (sql: string) => Effect.Effect + /** Run a SELECT and return its rows as plain records. */ + readonly query: (sql: string) => Effect.Effect>, MigrateSqlError> +} + +export class MigrationDriver extends Context.Service()( + "@maple-dev/effect-orm/MigrationDriver", +) {} + +/** The innermost message: drivers wrap the server's error in generic ones. */ +const firstLine = (cause: unknown): string => { + let current: unknown = cause + let message = String(cause) + for (let depth = 0; depth < 8 && typeof current === "object" && current !== null; depth++) { + if ("message" in current && typeof current.message === "string" && current.message.length > 0) message = current.message + const next: unknown = "reason" in current ? current.reason : "cause" in current ? current.cause : undefined + if (next === undefined || next === current) break + current = next + } + return message.split("\n")[0]?.trim().slice(0, 500) ?? message +} + +const sqlError = (sql: string) => (cause: unknown) => new MigrateSqlError({ message: firstLine(cause), sql, cause }) + +export interface FromSqlClientOptions { + /** + * Wraps statements that return no rows. Pass the ClickHouse client's + * `asCommand`; leave unset for drivers whose query path accepts DDL. + */ + readonly command?: (effect: Effect.Effect) => Effect.Effect +} + +/** A driver over an Effect `SqlClient`. */ +export const fromSqlClient = (sql: SqlClient.SqlClient, options: FromSqlClientOptions = {}): MigrationDriverApi => { + const command = options.command ?? ((effect) => effect) + return { + execute: (text) => command(sql.unsafe(text)).pipe(Effect.asVoid, Effect.mapError(sqlError(text))), + query: (text) => + sql.unsafe>(text).pipe( + Effect.map((rows): ReadonlyArray> => rows), + Effect.mapError(sqlError(text)), + ), + } +} + +/** A `MigrationDriver` layer over the `SqlClient` in context. */ +export const layerSqlClient = (options: FromSqlClientOptions = {}): Layer.Layer => + Layer.effect( + MigrationDriver, + Effect.gen(function* () { + return fromSqlClient(yield* SqlClient.SqlClient, options) + }), + ) diff --git a/src/migrate/errors.ts b/src/migrate/errors.ts new file mode 100644 index 0000000..2df7dec --- /dev/null +++ b/src/migrate/errors.ts @@ -0,0 +1,38 @@ +import { Schema } from "effect" + +/** A statement the server rejected, or a connection failure while running one. */ +export class MigrateSqlError extends Schema.TaggedError()("@maple-dev/effect-orm/MigrateSqlError", { + message: Schema.String, + sql: Schema.String, + cause: Schema.Defect(), +}) {} + +/** A migration file or snapshot that does not decode, or a migration set that cannot be ordered. */ +export class MigrateSourceError extends Schema.TaggedError()( + "@maple-dev/effect-orm/MigrateSourceError", + { migration: Schema.String, message: Schema.String, cause: Schema.optional(Schema.Defect()) }, +) {} + +/** An applied migration whose file has changed since it ran. */ +export class MigrateHashMismatch extends Schema.TaggedError()( + "@maple-dev/effect-orm/MigrateHashMismatch", + { migration: Schema.String, appliedHash: Schema.String, currentHash: Schema.String, message: Schema.String }, +) {} + +/** Another run holds the migration lease. */ +export class MigrateLeaseHeld extends Schema.TaggedError()("@maple-dev/effect-orm/MigrateLeaseHeld", { + owner: Schema.String, + expiresAt: Schema.String, + message: Schema.String, +}) {} + +/** A statement failed partway through a migration. Rerunning resumes at this step. */ +export class MigrateStepFailed extends Schema.TaggedError()("@maple-dev/effect-orm/MigrateStepFailed", { + migration: Schema.String, + step: Schema.String, + sql: Schema.String, + message: Schema.String, + cause: Schema.Defect(), +}) {} + +export type MigrateError = MigrateSqlError | MigrateSourceError | MigrateHashMismatch | MigrateLeaseHeld | MigrateStepFailed diff --git a/src/migrate/ledger.ts b/src/migrate/ledger.ts new file mode 100644 index 0000000..fdca6da --- /dev/null +++ b/src/migrate/ledger.ts @@ -0,0 +1,105 @@ +// The migration ledger, kept in the database being migrated. +// +// ClickHouse has no transactions and no unique keys, so the ledger is +// append-only and read with FINAL: +// +// - `_effect_orm_migrations`: one row per applied migration, written only +// after every statement in it finished. Never written first. +// - `_effect_orm_migration_steps`: one row per finished statement, so a run +// that stopped halfway resumes at the first statement without a row. +// - `_effect_orm_migration_lease`: who is migrating, until when. Best effort: +// two runs that start within the same instant can both see an empty lease. +// Callers that need a hard guarantee serialize runs themselves. + +import { Effect } from "effect" +import { quoteClickHouseString } from "../sql/sql-fragment" +import { ident, type RenderOptions } from "../schema/render" +import { MigrationDriver } from "./driver" +import type { MigrateSqlError } from "./errors" + +export const LEDGER_TABLES = { + migrations: "_effect_orm_migrations", + steps: "_effect_orm_migration_steps", + lease: "_effect_orm_migration_lease", +} as const + +const q = quoteClickHouseString + +const ledgerEngine = (options: RenderOptions, version: string): string => + options.replicated === undefined + ? `ReplacingMergeTree(${version})` + : `ReplicatedReplacingMergeTree(${q(options.replicated.path ?? "/clickhouse/tables/{shard}/{database}/{table}")}, ${q(options.replicated.replica ?? "{replica}")}, ${version})` + +const cluster = (options: RenderOptions): string => + options.cluster === undefined ? "" : ` ON CLUSTER ${ident(options.cluster)}` + +export const ensureLedger = (options: RenderOptions): Effect.Effect => + Effect.gen(function* () { + const driver = yield* MigrationDriver + yield* driver.execute( + `CREATE TABLE IF NOT EXISTS ${LEDGER_TABLES.migrations}${cluster(options)} (name String, hash String, applied_at DateTime64(3) DEFAULT now64(3)) ENGINE = ${ledgerEngine(options, "applied_at")} ORDER BY name`, + ) + yield* driver.execute( + `CREATE TABLE IF NOT EXISTS ${LEDGER_TABLES.steps}${cluster(options)} (name String, step String, sql_hash String, finished_at DateTime64(3) DEFAULT now64(3)) ENGINE = ${ledgerEngine(options, "finished_at")} ORDER BY (name, step)`, + ) + yield* driver.execute( + `CREATE TABLE IF NOT EXISTS ${LEDGER_TABLES.lease}${cluster(options)} (owner String, expires_at DateTime64(3), written_at DateTime64(3) DEFAULT now64(3)) ENGINE = ${ledgerEngine(options, "written_at")} ORDER BY owner`, + ) + }) + +export interface AppliedRow { + readonly name: string + readonly hash: string + readonly appliedAt: string +} + +export const readApplied = Effect.gen(function* () { + const driver = yield* MigrationDriver + const rows = yield* driver.query( + `SELECT name, hash, toString(applied_at) AS applied FROM ${LEDGER_TABLES.migrations} FINAL ORDER BY applied_at, name`, + ) + return rows.map((row): AppliedRow => ({ name: String(row.name), hash: String(row.hash), appliedAt: String(row.applied) })) +}) + +export const readDoneSteps = (name: string) => + Effect.gen(function* () { + const driver = yield* MigrationDriver + const rows = yield* driver.query(`SELECT step FROM ${LEDGER_TABLES.steps} FINAL WHERE name = ${q(name)}`) + return new Set(rows.map((row) => String(row.step))) + }) + +export const recordStep = (name: string, step: string, sqlHash: string) => + Effect.gen(function* () { + const driver = yield* MigrationDriver + yield* driver.execute( + `INSERT INTO ${LEDGER_TABLES.steps} (name, step, sql_hash) VALUES (${q(name)}, ${q(step)}, ${q(sqlHash)})`, + ) + }) + +export const recordMigration = (name: string, hash: string) => + Effect.gen(function* () { + const driver = yield* MigrationDriver + yield* driver.execute(`INSERT INTO ${LEDGER_TABLES.migrations} (name, hash) VALUES (${q(name)}, ${q(hash)})`) + }) + +export interface LeaseRow { + readonly owner: string + readonly expiresAt: string +} + +/** Leases that have not expired, oldest first. */ +export const readLiveLeases = Effect.gen(function* () { + const driver = yield* MigrationDriver + const rows = yield* driver.query( + `SELECT owner, toString(expires_at) AS expires FROM ${LEDGER_TABLES.lease} FINAL WHERE expires_at > now64(3) ORDER BY written_at, owner`, + ) + return rows.map((row): LeaseRow => ({ owner: String(row.owner), expiresAt: String(row.expires) })) +}) + +export const writeLease = (owner: string, seconds: number) => + Effect.gen(function* () { + const driver = yield* MigrationDriver + yield* driver.execute( + `INSERT INTO ${LEDGER_TABLES.lease} (owner, expires_at) VALUES (${q(owner)}, now64(3) + toIntervalSecond(${Math.max(0, Math.trunc(seconds))}))`, + ) + }) diff --git a/src/migrate/run.ts b/src/migrate/run.ts new file mode 100644 index 0000000..0660e24 --- /dev/null +++ b/src/migrate/run.ts @@ -0,0 +1,168 @@ +// Running migrations. +// +// Unlike Drizzle's migrator (one transaction, apply by name) and Effect's +// (insert the ledger rows first, then run, inside a transaction), this one +// assumes nothing is atomic. Each statement is journaled after it finishes and +// the migration row is written last, so a failure leaves an honest ledger and a +// rerun resumes at the statement that failed. + +import { Effect, Option } from "effect" +import { sha256Hex } from "../schema/entities" +import type { RenderOptions } from "../schema/render" +import { MigrationDriver } from "./driver" +import { MigrateHashMismatch, MigrateLeaseHeld, MigrateStepFailed, type MigrateError } from "./errors" +import { + ensureLedger, + readApplied, + readDoneSteps, + readLiveLeases, + recordMigration, + recordStep, + writeLease, + type AppliedRow, +} from "./ledger" +import { stepsOf, type LoadedMigration } from "./source" + +export interface RunOptions { + readonly migrations: ReadonlyArray + readonly render?: RenderOptions + /** Fail when an applied migration's file has changed. Otherwise log a warning. */ + readonly strict?: boolean + /** Names this run in the lease. Defaults to a random id. */ + readonly owner?: string + /** How long the lease lasts without renewal. Each finished statement renews it. */ + readonly leaseSeconds?: number +} + +export interface AppliedMigration { + readonly name: string + readonly steps: number + /** Steps skipped because an earlier run had finished them. */ + readonly resumedSteps: number +} + +export type MigrationState = "applied" | "pending" | "partial" | "changed" + +export interface MigrationStatus { + readonly name: string + readonly state: MigrationState + readonly appliedAt: string | undefined +} + +const checkHash = (migration: LoadedMigration, applied: AppliedRow, strict: boolean) => { + if (applied.hash === migration.hash) return Effect.void + const error = new MigrateHashMismatch({ + migration: migration.name, + appliedHash: applied.hash, + currentHash: migration.hash, + message: `${migration.name} changed after it was applied; write a new migration instead of editing this one`, + }) + return strict ? Effect.fail(error) : Effect.logWarning(error.message) +} + +const acquireLease = (owner: string, seconds: number) => + Effect.gen(function* () { + const others = (yield* readLiveLeases).filter((lease) => lease.owner !== owner) + const held = others[0] + if (held !== undefined) { + return yield* new MigrateLeaseHeld({ + owner: held.owner, + expiresAt: held.expiresAt, + message: `migrations are running as ${held.owner} until ${held.expiresAt}`, + }) + } + yield* writeLease(owner, seconds) + // Two runs that both saw no lease: the older write wins, the other backs off. + const first = (yield* readLiveLeases)[0] + if (first !== undefined && first.owner !== owner) { + yield* writeLease(owner, 0) + return yield* new MigrateLeaseHeld({ + owner: first.owner, + expiresAt: first.expiresAt, + message: `migrations are running as ${first.owner} until ${first.expiresAt}`, + }) + } + }) + +const applyOne = (migration: LoadedMigration, render: RenderOptions, renew: Effect.Effect) => + Effect.gen(function* () { + const driver = yield* MigrationDriver + const steps = stepsOf(migration, render) + const done = yield* readDoneSteps(migration.name) + let resumed = 0 + for (const step of steps) { + if (done.has(step.id)) { + resumed += 1 + continue + } + yield* driver.execute(step.sql).pipe( + Effect.mapError( + (cause) => + new MigrateStepFailed({ + migration: migration.name, + step: step.id, + sql: step.sql, + message: `${migration.name} step ${step.id} failed: ${cause.message}`, + cause, + }), + ), + Effect.withSpan("effect_orm.migrate.step", { attributes: { "effect_orm.migration.step": step.id } }), + ) + yield* recordStep(migration.name, step.id, yield* Effect.promise(() => sha256Hex(step.sql))) + yield* renew + } + yield* recordMigration(migration.name, migration.hash) + const result: AppliedMigration = { name: migration.name, steps: steps.length, resumedSteps: resumed } + return result + }).pipe( + Effect.withSpan("effect_orm.migrate.migration", { attributes: { "effect_orm.migration.name": migration.name } }), + ) + +/** + * Apply every migration not yet in the ledger, in order. Already-applied ones + * have their hash checked. Returns what ran. + */ +export const run = (options: RunOptions): Effect.Effect, MigrateError, MigrationDriver> => + Effect.gen(function* () { + const render = options.render ?? {} + const owner = options.owner ?? `effect-orm-${globalThis.crypto.randomUUID()}` + const leaseSeconds = options.leaseSeconds ?? 600 + yield* ensureLedger(render) + yield* acquireLease(owner, leaseSeconds) + const work = Effect.gen(function* () { + const applied = new Map((yield* readApplied).map((row) => [row.name, row])) + const ran: Array = [] + for (const migration of options.migrations) { + const row = applied.get(migration.name) + if (row !== undefined) { + yield* checkHash(migration, row, options.strict ?? false) + continue + } + ran.push(yield* applyOne(migration, render, writeLease(owner, leaseSeconds))) + } + return ran + }) + return yield* work.pipe(Effect.ensuring(writeLease(owner, 0).pipe(Effect.ignore))) + }).pipe(Effect.withSpan("effect_orm.migrate")) + +/** Where each migration stands, without running anything. */ +export const status = ( + migrations: ReadonlyArray, + render: RenderOptions = {}, +): Effect.Effect, MigrateError, MigrationDriver> => + Effect.gen(function* () { + yield* ensureLedger(render) + const applied = new Map((yield* readApplied).map((row) => [row.name, row])) + return yield* Effect.forEach(migrations, (migration) => + Effect.gen(function* () { + const row = Option.fromNullishOr(applied.get(migration.name)) + if (Option.isSome(row)) { + const state: MigrationState = row.value.hash === migration.hash ? "applied" : "changed" + return { name: migration.name, state, appliedAt: row.value.appliedAt } + } + const done = yield* readDoneSteps(migration.name) + const state: MigrationState = done.size > 0 ? "partial" : "pending" + return { name: migration.name, state, appliedAt: undefined } + }), + ) + }) diff --git a/src/migrate/source.ts b/src/migrate/source.ts new file mode 100644 index 0000000..e840989 --- /dev/null +++ b/src/migrate/source.ts @@ -0,0 +1,174 @@ +// Loading and ordering migrations. +// +// A migration is a folder name plus its file: `migration.json` (generated ops, +// rendered when it runs) or `migration.sql` (hand-written, statements split on +// `--> statement-breakpoint`, the drizzle-kit separator). Its `snapshot.json` +// is optional at run time; `verify` needs it, and ordering prefers it. + +import { Effect, FileSystem, Path, Schema } from "effect" +import { canonicalJson, sha256Hex, Snapshot } from "../schema/entities" +import { MigrationFile, renderOp } from "../schema/ops" +import type { RenderOptions } from "../schema/render" +import { MigrateSourceError } from "./errors" + +export const STATEMENT_BREAKPOINT = "--> statement-breakpoint" + +export interface MigrationInput { + /** `migration.json` or `migration.sql` contents. */ + readonly migration: string + readonly kind: "ops" | "sql" + /** `snapshot.json` contents, when present. */ + readonly snapshot?: string +} + +export interface LoadedMigration { + readonly name: string + readonly kind: "ops" | "sql" + /** sha256 of the migration file, recorded when it runs and checked afterwards. */ + readonly hash: string + readonly file: MigrationFile | undefined + readonly sql: ReadonlyArray + readonly snapshot: Snapshot | undefined +} + +export interface MigrationStep { + /** Stable within a migration: `.` or ``. */ + readonly id: string + readonly sql: string +} + +const decodeFile = Schema.decodeUnknownEffect(Schema.fromJsonString(MigrationFile)) +const decodeSnapshot = Schema.decodeUnknownEffect(Schema.fromJsonString(Snapshot)) + +const load = (name: string, input: MigrationInput): Effect.Effect => + Effect.gen(function* () { + const fail = (message: string) => (cause: unknown) => new MigrateSourceError({ migration: name, message, cause }) + const file = input.kind === "ops" ? yield* decodeFile(input.migration).pipe(Effect.mapError(fail("migration.json does not decode"))) : undefined + const snapshot = + input.snapshot === undefined + ? undefined + : yield* decodeSnapshot(input.snapshot).pipe(Effect.mapError(fail("snapshot.json does not decode"))) + // Hash the decoded content for ops, so reformatting the JSON is not an edit. + const hashed = file === undefined ? input.migration : canonicalJson(file) + const hash = yield* Effect.promise(() => sha256Hex(hashed)) + const sql = + input.kind === "sql" + ? input.migration + .split(STATEMENT_BREAKPOINT) + .map((statement) => statement.trim()) + .filter((statement) => statement.replace(/^\s*--.*$/gm, "").trim().length > 0) + : [] + return { name, kind: input.kind, hash, file, sql, snapshot } + }) + +/** + * Each migration's parent migrations, from its snapshot's `prevIds`. A custom + * migration copies its parent's entities, so several migrations can share one + * snapshot id; a parent link resolves to the last of them by name that sorts + * before the child. + */ +export const migrationParents = ( + migrations: ReadonlyArray, +): ReadonlyMap> => { + const byName = [...migrations].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)) + const bySnapshotId = new Map>() + for (const m of byName) { + if (m.snapshot === undefined) continue + const list = bySnapshotId.get(m.snapshot.id) ?? [] + list.push(m) + bySnapshotId.set(m.snapshot.id, list) + } + return new Map( + byName.map((m) => [ + m, + (m.snapshot?.prevIds ?? []).flatMap((id) => { + const holders = bySnapshotId.get(id)?.filter((h) => h !== m && h.name < m.name) ?? [] + return holders.length > 0 ? [holders.at(-1)!] : [] + }), + ]), + ) +} + +/** + * Order migrations by their snapshots' parent links, breaking ties between + * independent branches by name. Folder names alone are not enough: two + * migrations generated in the same second sort by their random suffix. + * Migrations without a snapshot keep their place by name. + */ +export const orderMigrations = ( + migrations: ReadonlyArray, +): Effect.Effect, MigrateSourceError> => + Effect.gen(function* () { + const byName = [...migrations].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)) + const names = new Set() + for (const m of byName) { + if (names.has(m.name)) return yield* new MigrateSourceError({ migration: m.name, message: "duplicate migration name" }) + names.add(m.name) + } + const parents = migrationParents(byName) + const parentsOf = (m: LoadedMigration): ReadonlyArray => parents.get(m) ?? [] + const ordered: Array = [] + const placed = new Set() + const visiting = new Set() + const visit = (m: LoadedMigration): boolean => { + if (placed.has(m)) return true + if (visiting.has(m)) return false + visiting.add(m) + for (const parent of parentsOf(m)) if (!visit(parent)) return false + visiting.delete(m) + placed.add(m) + ordered.push(m) + return true + } + for (const m of byName) { + if (!visit(m)) return yield* new MigrateSourceError({ migration: m.name, message: "snapshot parents form a cycle" }) + } + return ordered + }) + +/** Migrations from plain data: a bundler glob, an embedded record, a test. */ +export const fromRecord = ( + record: Readonly>, +): Effect.Effect, MigrateSourceError> => + Effect.forEach(Object.entries(record), ([name, input]) => load(name, input)).pipe(Effect.flatMap(orderMigrations)) + +/** Migrations from `//{migration.json | migration.sql, snapshot.json}`. */ +export const fromFileSystem = ( + directory: string, +): Effect.Effect, MigrateSourceError, FileSystem.FileSystem | Path.Path> => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem + const path = yield* Path.Path + const fail = (migration: string, message: string) => (cause: unknown) => + new MigrateSourceError({ migration, message, cause }) + const entries = yield* fs.readDirectory(directory).pipe(Effect.mapError(fail(directory, "cannot read the migrations directory"))) + const record: Record = {} + for (const name of entries) { + const dir = path.join(directory, name) + const read = (file: string) => + fs.exists(path.join(dir, file)).pipe( + Effect.flatMap((exists) => (exists ? Effect.map(fs.readFileString(path.join(dir, file)), (text): string | undefined => text) : Effect.succeed(undefined))), + Effect.mapError(fail(name, `cannot read ${file}`)), + ) + const json = yield* read("migration.json") + const sql = yield* read("migration.sql") + if (json !== undefined && sql !== undefined) { + return yield* new MigrateSourceError({ migration: name, message: "has both migration.json and migration.sql" }) + } + const migration = json ?? sql + if (migration === undefined) continue + const snapshot = yield* read("snapshot.json") + record[name] = { + migration, + kind: json !== undefined ? "ops" : "sql", + ...(snapshot !== undefined ? { snapshot } : undefined), + } + } + return yield* fromRecord(record) + }) + +/** The statements a migration runs, rendered for this deployment. */ +export const stepsOf = (migration: LoadedMigration, render: RenderOptions = {}): ReadonlyArray => + migration.file === undefined + ? migration.sql.map((sql, i) => ({ id: String(i), sql })) + : migration.file.ops.flatMap((op, i) => renderOp(op, render).map((sql, j) => ({ id: `${i}.${j}`, sql }))) diff --git a/src/migrate/verify.ts b/src/migrate/verify.ts new file mode 100644 index 0000000..c4768d8 --- /dev/null +++ b/src/migrate/verify.ts @@ -0,0 +1,208 @@ +// Drift: the live schema against the snapshot of the last applied migration. +// +// Not against the code's latest schema: a database three migrations behind is +// behind, not drifted. Expressions and types are normalized by the server +// (`formatQuery`, `defaultValueOfTypeName`), so `INTERVAL 30 DAY` and +// `toIntervalDay(30)`, or `DateTime64` and `DateTime64(3)`, compare equal. +// +// Checked: tables, engine family, sorting/partition/primary keys, columns +// (type, default), skipping indexes, views (target and body). Not checked yet: +// TTL, codecs, settings, comments. The server rewrites those in ways that need +// more than a formatter to compare. + +import { Effect } from "effect" +import { quoteClickHouseString } from "../sql/sql-fragment" +import type { ColumnEntity, IndexEntity, MaterializedViewEntity, Snapshot, TableEntity } from "../schema/entities" +import { MigrationDriver } from "./driver" +import type { MigrateSqlError } from "./errors" +import { LEDGER_TABLES, readApplied } from "./ledger" +import type { LoadedMigration } from "./source" + +export interface Drift { + readonly entity: string + readonly problem: + | "missing" + | "unexpected" + | "engine" + | "sorting_key" + | "partition_key" + | "primary_key" + | "type" + | "default" + | "index" + | "view_target" + | "view_body" + readonly expected?: string + readonly actual?: string +} + +export interface VerifyResult { + /** The migration whose snapshot was compared, if any applied migration has one. */ + readonly against: string | undefined + readonly drift: ReadonlyArray +} + +const q = quoteClickHouseString +const squash = (text: string): string => text.replace(/\s+/g, " ").trim() +const unwrapTuple = (key: string | null): string => { + if (key === null || key === "tuple()") return "" + const trimmed = key.trim() + return trimmed.startsWith("(") && trimmed.endsWith(")") ? trimmed.slice(1, -1) : trimmed +} +const engineFamily = (engine: string): string => engine.replace(/^(Replicated|Shared)(?=\w*MergeTree$)/, "") + +/** Server-normalized forms of SQL snippets, keyed by the input. */ +const canonical = (driver: typeof MigrationDriver.Service, kind: "expr" | "query", inputs: ReadonlySet) => + Effect.gen(function* () { + const list = [...inputs].filter((s) => s.length > 0) + const out = new Map() + if (list.length === 0) return out + const wrap = kind === "expr" ? "concat('SELECT ', x)" : "x" + const rows = yield* driver.query( + `SELECT arrayMap(x -> coalesce(formatQueryOrNull(${wrap}), x), [${list.map(q).join(", ")}]) AS out`, + ) + const formatted = rows[0]?.out + list.forEach((input, i) => out.set(input, squash(Array.isArray(formatted) ? String(formatted[i] ?? input) : input))) + return out + }) + +const canonicalTypes = (driver: typeof MigrationDriver.Service, types: ReadonlySet) => + Effect.gen(function* () { + const list = [...types] + const out = new Map() + for (const type of list) { + // One query per type: an unknown or argument-only type fails alone and keeps its text. + const rows = yield* driver + .query(`SELECT toTypeName(defaultValueOfTypeName(${q(type)})) AS t`) + .pipe(Effect.orElseSucceed(() => [{ t: type }])) + out.set(type, squash(String(rows[0]?.t ?? type))) + } + return out + }) + +export const verify = ( + migrations: ReadonlyArray, +): Effect.Effect => + Effect.gen(function* () { + const driver = yield* MigrationDriver + const appliedNames = new Set((yield* readApplied.pipe(Effect.orElseSucceed(() => []))).map((row) => row.name)) + const against = [...migrations].reverse().find((m) => appliedNames.has(m.name) && m.snapshot !== undefined) + const snapshot: Snapshot | undefined = against?.snapshot + if (snapshot === undefined) return { against: undefined, drift: [] } + + const db = String((yield* driver.query("SELECT currentDatabase() AS db"))[0]?.db ?? "default") + const tables = yield* driver.query( + `SELECT name, engine, sorting_key, partition_key, primary_key, as_select, create_table_query FROM system.tables WHERE database = currentDatabase() AND NOT is_temporary AND NOT startsWith(name, '.inner')`, + ) + const columns = yield* driver.query( + `SELECT table, name, type, default_kind, default_expression FROM system.columns WHERE database = currentDatabase()`, + ) + const indexes = yield* driver.query( + `SELECT table, name, type_full, expr, granularity FROM system.data_skipping_indices WHERE database = currentDatabase()`, + ) + + const expectedTables = snapshot.entities.filter((e): e is TableEntity => e.kind === "table") + const expectedColumns = snapshot.entities.filter((e): e is ColumnEntity => e.kind === "column") + const expectedIndexes = snapshot.entities.filter((e): e is IndexEntity => e.kind === "index") + const expectedViews = snapshot.entities.filter((e): e is MaterializedViewEntity => e.kind === "materialized_view") + + const exprs = new Set() + for (const t of expectedTables) for (const k of [t.orderBy, t.partitionBy, t.primaryKey]) exprs.add(unwrapTuple(k)) + for (const c of expectedColumns) if (c.default !== null) exprs.add(c.default.expr) + for (const i of expectedIndexes) exprs.add(i.expr) + for (const row of tables) for (const k of ["sorting_key", "partition_key", "primary_key"]) exprs.add(String(row[k] ?? "")) + for (const row of columns) exprs.add(String(row.default_expression ?? "")) + for (const row of indexes) exprs.add(String(row.expr ?? "")) + const expr = yield* canonical(driver, "expr", exprs) + const qualified = new RegExp(`\\b${db.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}\\.`, "g") + const views = new Set(expectedViews.map((v) => v.select)) + for (const row of tables) if (row.engine === "MaterializedView") views.add(String(row.as_select ?? "").replace(qualified, "")) + const query = yield* canonical(driver, "query", views) + const types = yield* canonicalTypes(driver, new Set([...expectedColumns.map((c) => c.type), ...columns.map((c) => String(c.type))])) + + const norm = (map: Map, value: string): string => (value.length === 0 ? "" : (map.get(value) ?? squash(value))) + const drift: Array = [] + const actualTables = new Map(tables.map((row) => [String(row.name), row])) + const managed = new Set(Object.values(LEDGER_TABLES)) + + for (const t of expectedTables) { + const row = actualTables.get(t.name) + if (row === undefined) { + drift.push({ entity: t.name, problem: "missing" }) + continue + } + if (engineFamily(String(row.engine)) !== t.engine.family) { + drift.push({ entity: t.name, problem: "engine", expected: t.engine.family, actual: String(row.engine) }) + } + const keys = [ + ["sorting_key", t.orderBy], + ["partition_key", t.partitionBy], + ["primary_key", t.primaryKey ?? t.orderBy], + ] as const + for (const [column, key] of keys) { + const expected = norm(expr, unwrapTuple(key)) + const actual = norm(expr, String(row[column] ?? "")) + if (expected !== actual) drift.push({ entity: t.name, problem: column, expected, actual }) + } + } + for (const v of expectedViews) { + const row = actualTables.get(v.name) + if (row === undefined) { + drift.push({ entity: v.name, problem: "missing" }) + continue + } + const target = /\bTO\s+(\S+)/.exec(String(row.create_table_query ?? ""))?.[1]?.replace(qualified, "").replace(/`/g, "") + if (target !== v.to) drift.push({ entity: v.name, problem: "view_target", expected: v.to, ...(target !== undefined ? { actual: target } : undefined) }) + const expected = norm(query, v.select) + const actual = norm(query, String(row.as_select ?? "").replace(qualified, "")) + if (expected !== actual) drift.push({ entity: v.name, problem: "view_body", expected, actual }) + } + const expectedNames = new Set([...expectedTables.map((t) => t.name), ...expectedViews.map((v) => v.name)]) + for (const name of actualTables.keys()) { + if (!expectedNames.has(name) && !managed.has(name)) drift.push({ entity: name, problem: "unexpected" }) + } + + const actualColumns = new Map(columns.map((row) => [`${String(row.table)}.${String(row.name)}`, row])) + const tableNames = new Set(expectedTables.map((t) => t.name)) + for (const c of expectedColumns) { + const key = `${c.table}.${c.name}` + const row = actualColumns.get(key) + if (row === undefined) { + if (actualTables.has(c.table)) drift.push({ entity: key, problem: "missing" }) + continue + } + const expectedType = types.get(c.type) ?? c.type + const actualType = types.get(String(row.type)) ?? String(row.type) + if (expectedType !== actualType) drift.push({ entity: key, problem: "type", expected: expectedType, actual: actualType }) + const expectedDefault = c.default === null ? "" : `${c.default.kind} ${norm(expr, c.default.expr)}` + const actualKind = String(row.default_kind ?? "") + const actualDefault = actualKind.length === 0 ? "" : `${actualKind} ${norm(expr, String(row.default_expression ?? ""))}` + if (expectedDefault !== actualDefault) drift.push({ entity: key, problem: "default", expected: expectedDefault, actual: actualDefault }) + } + for (const [key, row] of actualColumns) { + const table = String(row.table) + if (tableNames.has(table) && !expectedColumns.some((c) => `${c.table}.${c.name}` === key)) { + drift.push({ entity: key, problem: "unexpected" }) + } + } + + const actualIndexes = new Map(indexes.map((row) => [`${String(row.table)}.${String(row.name)}`, row])) + for (const i of expectedIndexes) { + const key = `${i.table}.${i.name}` + const row = actualIndexes.get(key) + if (row === undefined) { + if (actualTables.has(i.table)) drift.push({ entity: key, problem: "missing" }) + continue + } + const expected = `${norm(expr, i.expr)} TYPE ${squash(i.type)} GRANULARITY ${i.granularity}` + const actual = `${norm(expr, String(row.expr ?? ""))} TYPE ${squash(String(row.type_full ?? ""))} GRANULARITY ${String(row.granularity ?? "")}` + if (expected !== actual) drift.push({ entity: key, problem: "index", expected, actual }) + } + for (const [key, row] of actualIndexes) { + if (tableNames.has(String(row.table)) && !expectedIndexes.some((i) => `${i.table}.${i.name}` === key)) { + drift.push({ entity: key, problem: "unexpected" }) + } + } + + return { against: against?.name, drift } + }) diff --git a/src/schema.ts b/src/schema.ts new file mode 100644 index 0000000..1677959 --- /dev/null +++ b/src/schema.ts @@ -0,0 +1,58 @@ +// @maple-dev/effect-orm/schema +// +// Tables and materialized views that carry their DDL, snapshots of them, and +// the offline diff that turns two snapshots into migration ops. Pure: nothing +// here reads files or opens a connection. See docs/migrations.md. + +export { + column, + defineTable, + engine, + index, + materializedView, + ttlAfterDays, + SchemaDefinitionDefect, + type ColumnInput, + type ColumnOptions, + type ColumnSpec, + type ColumnsOf, + type DdlExpr, + type DdlKey, + type IndexSpec, + type MaterializedView, + type MisfitColumns, + type SchemaTable, + type TableDdl, + type TableDefinition, +} from "./schema/define" +export { + ColumnDefault, + ColumnEntity, + EngineSpec, + IndexEntity, + MaterializedViewEntity, + ORIGIN_ID, + SchemaEntity, + Snapshot, + SNAPSHOT_VERSION, + TableEntity, + canonicalJson, + entityKey, + sha256Hex, + sortEntities, +} from "./schema/entities" +export { + ident, + renderAlter, + renderColumnDefinition, + renderCreateMaterializedView, + renderCreateTable, + renderDropTable, + renderDropView, + renderEngine, + renderSchema, + type RenderOptions, +} from "./schema/render" +export { MigrationFile, MigrationOp, labelOf, renderOp, type OpLabel } from "./schema/ops" +export { entitiesOf, isSchemaObject, makeSnapshot, serializeSnapshot, type SchemaObject } from "./schema/snapshot" +export { Hint, Hints, diffSchemas, type DiffResult, type UnsupportedChange } from "./schema/diff" diff --git a/src/schema/define.ts b/src/schema/define.ts new file mode 100644 index 0000000..205a84b --- /dev/null +++ b/src/schema/define.ts @@ -0,0 +1,364 @@ +// Schema definitions: tables and materialized views that carry their DDL. +// +// `defineTable` returns a value that IS a `Table`, so every query API accepts it +// unchanged; the DDL rides beside it on `ddl`, already normalized into +// entities. Expressions (keys, TTL, defaults, index and view bodies) are +// written with the query DSL and rendered once, here, with the ClickHouse +// dialect. A definition that cannot be rendered is a bug in the definition, so +// it dies as a `SchemaDefinitionDefect` at module load rather than surfacing +// later as a failed migration. + +import { Schema } from "effect" +import { compileCHUnsafe } from "../ch/compile" +import { clickhouseDialect, withDialect } from "../ch/dialect" +import type { Expr } from "../ch/expr" +import { encodeColumnLiteral } from "../ch/literal" +import { createColumnAccessor, type CHQuery, type ColumnAccessor } from "../ch/query" +import type { Table } from "../ch/table" +import type { CHType, ColumnDefs, InferTS } from "../ch/types" +import { compile as compileFragment } from "../sql/sql-fragment" +import type { + ColumnDefault, + ColumnEntity, + EngineSpec, + IndexEntity, + MaterializedViewEntity, + TableEntity, +} from "./entities" + +/** A schema definition that cannot be turned into DDL. Raised while the module loads. */ +export class SchemaDefinitionDefect extends Schema.TaggedError()( + "@maple-dev/effect-orm/SchemaDefinitionDefect", + { object: Schema.String, message: Schema.String }, +) {} + +// Expressions + +/** SQL written as a string, or built with the DSL against the table's columns. */ +export type DdlExpr = string | (($: ColumnAccessor) => Expr | string) + +/** Column names, or a callback returning expressions. Renders as a key tuple. */ +export type DdlKey = + | ReadonlyArray + | (($: ColumnAccessor) => ReadonlyArray | string>) + +const renderExprValue = (value: Expr | string): string => + typeof value === "string" ? value : withDialect(clickhouseDialect, () => compileFragment(value.toFragment())) + +const renderExpr = (expr: DdlExpr, columns: Cols): string => + typeof expr === "string" ? expr : renderExprValue(expr(createColumnAccessor(columns))) + +const renderKey = (key: DdlKey, columns: Cols): string => { + const parts = typeof key === "function" ? key(createColumnAccessor(columns)).map(renderExprValue) : [...key] + return parts.length === 1 ? parts[0]! : `(${parts.join(", ")})` +} + +// Columns + +export interface ColumnOptions> { + /** A literal default, encoded through the column's own type. */ + readonly default?: InferTS + /** `DEFAULT `, for a computed default. */ + readonly defaultExpr?: DdlExpr + /** `MATERIALIZED `: computed on insert, never inserted directly. */ + readonly materialized?: DdlExpr + /** `ALIAS `: computed on read, not stored. */ + readonly alias?: DdlExpr + /** Compression codec, written as after `CODEC`, e.g. `"Delta, ZSTD(1)"`. */ + readonly codec?: string + readonly comment?: string +} + +export interface ColumnSpec> { + readonly _tag: "ColumnSpec" + readonly type: T + readonly options: ColumnOptions +} + +/** A column with DDL options. A bare column type works too where no option is needed. */ +export const column = >(type: T, options: ColumnOptions = {}): ColumnSpec => ({ + _tag: "ColumnSpec", + type, + options, +}) + +export type ColumnInput = CHType | ColumnSpec> + +/** The query-side column types of a `columns` record. */ +export type ColumnsOf> = { + readonly [K in keyof I]: I[K] extends ColumnSpec + ? T + : I[K] extends CHType + ? I[K] + : never +} + +const isColumnSpec = (input: ColumnInput): input is ColumnSpec> => + "_tag" in input && input._tag === "ColumnSpec" + +// Engines + +const engineOf = (family: string, ...params: ReadonlyArray): EngineSpec => ({ + family, + params: params.filter((p): p is string => p !== undefined), +}) + +/** + * Table engines. Write the plain family: the `Replicated` prefix and its + * Keeper path are a render option, so one schema serves a single server, a + * replicated cluster, and ClickHouse Cloud (which converts MergeTree itself). + */ +export const engine = { + mergeTree: (): EngineSpec => engineOf("MergeTree"), + replacingMergeTree: (options: { readonly version?: string; readonly isDeleted?: string } = {}): EngineSpec => + engineOf("ReplacingMergeTree", options.version, options.version ? options.isDeleted : undefined), + summingMergeTree: (options: { readonly columns?: ReadonlyArray } = {}): EngineSpec => + engineOf("SummingMergeTree", options.columns ? `(${options.columns.join(", ")})` : undefined), + aggregatingMergeTree: (): EngineSpec => engineOf("AggregatingMergeTree"), + collapsingMergeTree: (sign: string): EngineSpec => engineOf("CollapsingMergeTree", sign), + versionedCollapsingMergeTree: (sign: string, version: string): EngineSpec => + engineOf("VersionedCollapsingMergeTree", sign, version), + null: (): EngineSpec => engineOf("Null"), + memory: (): EngineSpec => engineOf("Memory"), +} as const + +const isMergeTreeFamily = (spec: EngineSpec): boolean => spec.family.endsWith("MergeTree") + +// Indexes and TTL + +export interface IndexSpec { + readonly name: string + readonly expr: DdlExpr + /** The index type as written after `TYPE`, e.g. `minmax`, `bloom_filter(0.01)`, `set(100)`. */ + readonly type: string + readonly granularity?: number +} + +/** A data-skipping index. */ +export const index = ( + name: string, + expr: DdlExpr, + type: string, + granularity = 1, +): IndexSpec => ({ name, expr, type, granularity }) + +/** ` + INTERVAL n DAY`, the usual row TTL. */ +export const ttlAfterDays = + (expr: DdlExpr, days: number) => + ($: ColumnAccessor): string => + `${typeof expr === "string" ? expr : renderExprValue(expr($))} + toIntervalDay(${Math.trunc(days)})` + +// Tables + +export interface TableDefinition> { + readonly columns: Columns + readonly engine: EngineSpec + /** Required for the MergeTree family; use `[]` for `ORDER BY tuple()`. */ + readonly orderBy?: DdlKey> + readonly partitionBy?: DdlExpr> + readonly primaryKey?: DdlKey> + readonly ttl?: DdlExpr> + readonly settings?: Readonly> + readonly indexes?: ReadonlyArray>> + readonly comment?: string + /** As for `table()`: the column carrying row-level tenancy. */ + readonly tenantColumn?: keyof Columns & string +} + +/** The DDL a `defineTable` value carries, as entities. */ +export interface TableDdl { + readonly table: TableEntity + readonly columns: ReadonlyArray + readonly indexes: ReadonlyArray +} + +export interface SchemaTable extends Table { + readonly ddl: TableDdl +} + +const IDENTIFIER = /^[A-Za-z_][A-Za-z0-9_]*$/ + +const assertIdentifier = (object: string, name: string): void => { + if (!IDENTIFIER.test(name)) { + throw new SchemaDefinitionDefect({ + object, + message: `${JSON.stringify(name)} is not a plain identifier ([A-Za-z_][A-Za-z0-9_]*)`, + }) + } +} + +const columnDefault = ( + table: string, + name: string, + spec: ColumnSpec>, + columns: ColumnDefs, +): ColumnDefault | null => { + const { options } = spec + const set = [ + options.default !== undefined, + options.defaultExpr !== undefined, + options.materialized !== undefined, + options.alias !== undefined, + ].filter(Boolean).length + if (set > 1) { + throw new SchemaDefinitionDefect({ + object: `${table}.${name}`, + message: "a column takes at most one of default, defaultExpr, materialized, alias", + }) + } + if (options.default !== undefined) { + return { + kind: "DEFAULT", + expr: withDialect(clickhouseDialect, () => encodeColumnLiteral(spec.type, options.default, name)), + } + } + if (options.defaultExpr !== undefined) return { kind: "DEFAULT", expr: renderExpr(options.defaultExpr, columns) } + if (options.materialized !== undefined) { + return { kind: "MATERIALIZED", expr: renderExpr(options.materialized, columns) } + } + if (options.alias !== undefined) return { kind: "ALIAS", expr: renderExpr(options.alias, columns) } + return null +} + +/** + * A table with its DDL. Usable everywhere a `table()` is; `generate` reads its + * `ddl` to produce migrations. + */ +export function defineTable>( + name: Name, + definition: TableDefinition, +): SchemaTable> { + assertIdentifier(name, name) + const inputs = Object.entries(definition.columns) + if (inputs.length === 0) throw new SchemaDefinitionDefect({ object: name, message: "a table needs columns" }) + const types = Object.fromEntries( + inputs.map(([column, input]) => [column, isColumnSpec(input) ? input.type : input]), + ) as ColumnsOf + + if (isMergeTreeFamily(definition.engine) && definition.orderBy === undefined) { + throw new SchemaDefinitionDefect({ + object: name, + message: `${definition.engine.family} needs orderBy (use [] for ORDER BY tuple())`, + }) + } + if (!isMergeTreeFamily(definition.engine)) { + const misplaced = (["orderBy", "partitionBy", "primaryKey", "ttl", "indexes"] as const).filter( + (key) => definition[key] !== undefined, + ) + if (misplaced.length > 0) { + throw new SchemaDefinitionDefect({ + object: name, + message: `${definition.engine.family} takes no ${misplaced.join(", ")}`, + }) + } + } + + const columnEntities = inputs.map(([column, input], position): ColumnEntity => { + assertIdentifier(`${name}.${column}`, column) + const spec = isColumnSpec(input) ? input : { _tag: "ColumnSpec" as const, type: input, options: {} } + return { + kind: "column", + table: name, + name: column, + position, + type: spec.type.sql, + default: columnDefault(name, column, spec, types), + codec: spec.options.codec ?? null, + comment: spec.options.comment ?? null, + } + }) + + const orderBy = + definition.orderBy === undefined + ? null + : Array.isArray(definition.orderBy) && definition.orderBy.length === 0 + ? "tuple()" + : renderKey(definition.orderBy, types) + + const indexEntities = (definition.indexes ?? []).map((spec): IndexEntity => { + assertIdentifier(`${name} index`, spec.name) + return { + kind: "index", + table: name, + name: spec.name, + expr: renderExpr(spec.expr, types), + type: spec.type, + granularity: spec.granularity ?? 1, + } + }) + + const table: TableEntity = { + kind: "table", + name, + engine: definition.engine, + orderBy, + partitionBy: definition.partitionBy === undefined ? null : renderExpr(definition.partitionBy, types), + primaryKey: definition.primaryKey === undefined ? null : renderKey(definition.primaryKey, types), + ttl: definition.ttl === undefined ? null : renderExpr(definition.ttl, types), + settings: Object.fromEntries(Object.entries(definition.settings ?? {}).map(([k, v]) => [k, String(v)])), + comment: definition.comment ?? null, + } + + return { + _tag: "Table", + name, + columns: types, + ...(definition.tenantColumn !== undefined ? { tenantColumn: definition.tenantColumn } : undefined), + ddl: { table, columns: columnEntities, indexes: indexEntities }, + } +} + +// Materialized views + +/** Output columns the target cannot take: missing there, or a different type. */ +export type MisfitColumns = { + [K in keyof Output]: K extends keyof Cols ? ([Output[K]] extends [InferTS] ? never : K) : K +}[keyof Output] + +export interface MaterializedView { + readonly _tag: "MaterializedView" + readonly name: Name + readonly ddl: MaterializedViewEntity +} + +const leftmostTable = (query: CHQuery): string => { + const state = query._state + if (state.fromQuery !== undefined) return leftmostTable(state.fromQuery) + if (state.fromUnion !== undefined) { + throw new SchemaDefinitionDefect({ + object: state.tableName, + message: "a materialized view cannot read FROM a union; define one view per branch", + }) + } + return state.tableName +} + +/** + * A materialized view writing to `to`. Its body is a DSL query, so its output + * is checked against the target's columns: a column the target lacks, or of + * another type, is a type error here instead of a failed insert later. + * + * Views never `POPULATE`. History comes from an explicit backfill. + */ +export function materializedView< + const Name extends string, + TargetName extends string, + Cols extends ColumnDefs, + Output extends Record, +>( + name: Name, + options: { + readonly to: SchemaTable + readonly as: CHQuery + } & ([MisfitColumns] extends [never] + ? unknown + : { readonly targetCannotTake: MisfitColumns }), +): MaterializedView { + assertIdentifier(name, name) + const select = compileCHUnsafe(options.as, {}, { skipFormat: true, dialect: clickhouseDialect }).sql + return { + _tag: "MaterializedView", + name, + ddl: { kind: "materialized_view", name, to: options.to.name, sources: [leftmostTable(options.as)], select }, + } +} diff --git a/src/schema/diff.ts b/src/schema/diff.ts new file mode 100644 index 0000000..f1799c1 --- /dev/null +++ b/src/schema/diff.ts @@ -0,0 +1,228 @@ +// Diff two schemas into migration ops. +// +// Pure and offline, like drizzle-kit's `generate`. Two things stop it from +// writing a migration on its own: +// +// - Data loss (dropping a table or a column) needs a `confirm_data_loss` hint, +// from a prompt or from `--hints`. Without one the change is reported as a +// missing hint and nothing is written, the way drizzle-kit exits 2. +// - Changes ClickHouse cannot make with ALTER (engine, sorting key, partition +// key, primary key, column type) are reported as unsupported. They need a +// table rebuild, which this version does not generate. +// +// Renames are not detected yet: a renamed column reads as a drop plus an add, +// and the drop asks for confirmation, so a rename never loses data silently. + +import { Schema } from "effect" +import { + entityKey, + type ColumnEntity, + type IndexEntity, + type MaterializedViewEntity, + type SchemaEntity, + type TableEntity, +} from "./entities" +import type { MigrationOp } from "./ops" + +export const Hint = Schema.Union([ + Schema.Struct({ + type: Schema.Literal("confirm_data_loss"), + kind: Schema.Literals(["table", "column"]), + /** `table` or `table.column`. */ + entity: Schema.String, + }), +]) +export type Hint = typeof Hint.Type + +export const Hints = Schema.Array(Hint) + +export interface UnsupportedChange { + readonly entity: string + readonly message: string +} + +export interface DiffResult { + readonly ops: ReadonlyArray + readonly missingHints: ReadonlyArray + readonly unsupported: ReadonlyArray +} + +const hintKey = (hint: Hint): string => `${hint.type}:${hint.kind}:${hint.entity}` + +const byKind = (entities: ReadonlyArray, kind: K) => + new Map( + entities + .filter((e): e is Extract => e.kind === kind) + .map((e) => [entityKey(e), e] as const), + ) + +const same = (a: unknown, b: unknown): boolean => JSON.stringify(a) === JSON.stringify(b) + +const opOrder: Record = { + drop_view: 0, + drop_index: 1, + drop_column: 2, + drop_table: 3, + create_table: 4, + add_column: 5, + modify_column: 6, + modify_ttl: 7, + modify_settings: 8, + modify_comment: 9, + add_index: 10, + create_view: 11, +} + +export const diffSchemas = ( + prev: ReadonlyArray, + next: ReadonlyArray, + hints: ReadonlyArray = [], +): DiffResult => { + const given = new Set(hints.map(hintKey)) + const ops: Array = [] + const missingHints: Array = [] + const unsupported: Array = [] + + const confirm = (hint: Hint, op: MigrationOp): void => { + if (given.has(hintKey(hint))) ops.push(op) + else missingHints.push(hint) + } + + const prevTables = byKind(prev, "table") + const nextTables = byKind(next, "table") + const prevColumns = byKind(prev, "column") + const nextColumns = byKind(next, "column") + const prevIndexes = byKind(prev, "index") + const nextIndexes = byKind(next, "index") + + const columnsOf = (columns: Map, table: string): ReadonlyArray => + [...columns.values()].filter((c) => c.table === table).sort((a, b) => a.position - b.position) + const indexesOf = (indexes: Map, table: string): ReadonlyArray => + [...indexes.values()].filter((i) => i.table === table) + + for (const [key, table] of nextTables) { + if (!prevTables.has(key)) { + ops.push({ + op: "create_table", + table, + columns: columnsOf(nextColumns, table.name), + indexes: indexesOf(nextIndexes, table.name), + }) + } + } + for (const [key, table] of prevTables) { + if (!nextTables.has(key)) { + confirm({ type: "confirm_data_loss", kind: "table", entity: table.name }, { op: "drop_table", name: table.name }) + } + } + + for (const [key, after] of nextTables) { + const before = prevTables.get(key) + if (before === undefined) continue + diffTable(before, after, ops, unsupported) + + const beforeColumns = columnsOf(prevColumns, after.name) + const afterColumns = columnsOf(nextColumns, after.name) + const beforeByName = new Map(beforeColumns.map((c) => [c.name, c])) + const afterNames = new Set(afterColumns.map((c) => c.name)) + + afterColumns.forEach((column, i) => { + const old = beforeByName.get(column.name) + if (old === undefined) { + ops.push({ op: "add_column", column, after: i === 0 ? null : afterColumns[i - 1]!.name }) + return + } + if (old.type !== column.type) { + unsupported.push({ + entity: `${after.name}.${column.name}`, + message: `type ${old.type} -> ${column.type} rewrites data and is not generated yet`, + }) + return + } + if (!same(old.default, column.default) || old.codec !== column.codec || old.comment !== column.comment) { + ops.push({ op: "modify_column", from: old, to: column }) + } + }) + for (const column of beforeColumns) { + if (!afterNames.has(column.name)) { + confirm( + { type: "confirm_data_loss", kind: "column", entity: `${after.name}.${column.name}` }, + { op: "drop_column", table: after.name, name: column.name }, + ) + } + } + + const beforeIndexes = new Map(indexesOf(prevIndexes, after.name).map((i) => [i.name, i])) + const afterIndexes = indexesOf(nextIndexes, after.name) + for (const index of afterIndexes) { + const old = beforeIndexes.get(index.name) + if (old !== undefined && same(old, index)) continue + if (old !== undefined) ops.push({ op: "drop_index", table: after.name, name: index.name }) + ops.push({ op: "add_index", index }) + } + for (const old of beforeIndexes.values()) { + if (!afterIndexes.some((i) => i.name === old.name)) { + ops.push({ op: "drop_index", table: after.name, name: old.name }) + } + } + } + + diffViews(byKind(prev, "materialized_view"), byKind(next, "materialized_view"), ops) + + // Dropping a table a kept view still reads would leave that view failing every insert. + const droppedTables = new Set(ops.flatMap((op) => (op.op === "drop_table" ? [op.name] : []))) + for (const view of byKind(next, "materialized_view").values()) { + for (const source of view.sources) { + if (droppedTables.has(source)) { + unsupported.push({ entity: view.name, message: `reads ${source}, which this migration drops` }) + } + } + } + + return { + ops: ops.map((op, i) => [op, i] as const).sort(([a, i], [b, j]) => opOrder[a.op] - opOrder[b.op] || i - j).map(([op]) => op), + missingHints, + unsupported, + } +} + +const diffTable = ( + before: TableEntity, + after: TableEntity, + ops: Array, + unsupported: Array, +): void => { + const rebuild = (what: string, from: unknown, to: unknown) => + unsupported.push({ + entity: after.name, + message: `${what} ${JSON.stringify(from)} -> ${JSON.stringify(to)} needs a table rebuild, which is not generated yet`, + }) + if (!same(before.engine, after.engine)) rebuild("engine", before.engine, after.engine) + if (before.orderBy !== after.orderBy) rebuild("ORDER BY", before.orderBy, after.orderBy) + if (before.partitionBy !== after.partitionBy) rebuild("PARTITION BY", before.partitionBy, after.partitionBy) + if (before.primaryKey !== after.primaryKey) rebuild("PRIMARY KEY", before.primaryKey, after.primaryKey) + if (before.ttl !== after.ttl) ops.push({ op: "modify_ttl", table: after.name, ttl: after.ttl }) + if (!same(before.settings, after.settings)) { + const set = Object.fromEntries(Object.entries(after.settings).filter(([k, v]) => before.settings[k] !== v)) + const reset = Object.keys(before.settings).filter((k) => !(k in after.settings)) + ops.push({ op: "modify_settings", table: after.name, set, reset }) + } + if (before.comment !== after.comment) ops.push({ op: "modify_comment", table: after.name, comment: after.comment }) +} + +/** A view's body is frozen at creation, so any change is a drop and a re-create. */ +const diffViews = ( + before: Map, + after: Map, + ops: Array, +): void => { + for (const [key, view] of after) { + const old = before.get(key) + if (old !== undefined && same(old, view)) continue + if (old !== undefined) ops.push({ op: "drop_view", name: view.name }) + ops.push({ op: "create_view", view }) + } + for (const [key, view] of before) { + if (!after.has(key)) ops.push({ op: "drop_view", name: view.name }) + } +} diff --git a/src/schema/entities.ts b/src/schema/entities.ts new file mode 100644 index 0000000..9c18f80 --- /dev/null +++ b/src/schema/entities.ts @@ -0,0 +1,132 @@ +// Schema entities: the normalized, serializable form of a ClickHouse schema. +// +// A `defineTable` value is code; a snapshot is data. Everything downstream of +// the definitions (DDL rendering, diffing, the migrator's drift check) reads +// these entities, never the definitions, so a snapshot taken months ago renders +// and diffs exactly as it did when it was written. + +import { Schema } from "effect" + +/** `DEFAULT`, `MATERIALIZED`, or `ALIAS`, with its SQL expression. */ +export const ColumnDefault = Schema.Struct({ + kind: Schema.Literals(["DEFAULT", "MATERIALIZED", "ALIAS"]), + expr: Schema.String, +}) +export type ColumnDefault = typeof ColumnDefault.Type + +export const ColumnEntity = Schema.Struct({ + kind: Schema.Literal("column"), + table: Schema.String, + name: Schema.String, + /** Declaration order, which `ADD COLUMN ... AFTER` preserves. */ + position: Schema.Number, + type: Schema.String, + default: Schema.NullOr(ColumnDefault), + codec: Schema.NullOr(Schema.String), + comment: Schema.NullOr(Schema.String), +}) +export type ColumnEntity = typeof ColumnEntity.Type + +export const EngineSpec = Schema.Struct({ + /** The MergeTree family member without a `Replicated` prefix, or `Null` / `Memory`. */ + family: Schema.String, + /** Engine arguments as SQL, e.g. `["Version"]` for `ReplacingMergeTree(Version)`. */ + params: Schema.Array(Schema.String), +}) +export type EngineSpec = typeof EngineSpec.Type + +export const TableEntity = Schema.Struct({ + kind: Schema.Literal("table"), + name: Schema.String, + engine: EngineSpec, + orderBy: Schema.NullOr(Schema.String), + partitionBy: Schema.NullOr(Schema.String), + primaryKey: Schema.NullOr(Schema.String), + ttl: Schema.NullOr(Schema.String), + settings: Schema.Record(Schema.String, Schema.String), + comment: Schema.NullOr(Schema.String), +}) +export type TableEntity = typeof TableEntity.Type + +export const IndexEntity = Schema.Struct({ + kind: Schema.Literal("index"), + table: Schema.String, + name: Schema.String, + expr: Schema.String, + type: Schema.String, + granularity: Schema.Number, +}) +export type IndexEntity = typeof IndexEntity.Type + +export const MaterializedViewEntity = Schema.Struct({ + kind: Schema.Literal("materialized_view"), + name: Schema.String, + /** Target table the view writes to (`TO `). */ + to: Schema.String, + /** Tables the body reads; a write to any of them runs the view. */ + sources: Schema.Array(Schema.String), + select: Schema.String, +}) +export type MaterializedViewEntity = typeof MaterializedViewEntity.Type + +export const SchemaEntity = Schema.Union([TableEntity, ColumnEntity, IndexEntity, MaterializedViewEntity]) +export type SchemaEntity = typeof SchemaEntity.Type + +/** The snapshot format version this build writes and reads. */ +export const SNAPSHOT_VERSION = "1" + +/** The parent id of a first migration. */ +export const ORIGIN_ID = "0000000000000000000000000000000000000000000000000000000000000000" + +export const Snapshot = Schema.Struct({ + version: Schema.Literal(SNAPSHOT_VERSION), + dialect: Schema.Literal("clickhouse"), + /** sha256 of the canonical entity list; two branches reaching one schema agree. */ + id: Schema.String, + prevIds: Schema.Array(Schema.String), + entities: Schema.Array(SchemaEntity), +}) +export type Snapshot = typeof Snapshot.Type + +/** A stable key per entity, unique within a snapshot. */ +export const entityKey = (entity: SchemaEntity): string => { + switch (entity.kind) { + case "table": + return `table:${entity.name}` + case "materialized_view": + return `materialized_view:${entity.name}` + case "column": + return `column:${entity.table}.${entity.name}` + case "index": + return `index:${entity.table}.${entity.name}` + } +} + +const kindOrder: Record = { table: 0, column: 1, index: 2, materialized_view: 3 } + +/** Tables, then columns in declaration order, then indexes, then views. Deterministic. */ +export const sortEntities = (entities: ReadonlyArray): ReadonlyArray => + [...entities].sort((a, b) => { + const byKind = kindOrder[a.kind] - kindOrder[b.kind] + if (byKind !== 0) return byKind + if (a.kind === "column" && b.kind === "column") { + return a.table === b.table ? a.position - b.position : a.table < b.table ? -1 : 1 + } + const ka = entityKey(a) + const kb = entityKey(b) + return ka < kb ? -1 : ka > kb ? 1 : 0 + }) + +/** JSON with object keys sorted, so equal values serialize to equal bytes. */ +export const canonicalJson = (value: unknown): string => + JSON.stringify(value, (_key, v: unknown) => + v !== null && typeof v === "object" && !Array.isArray(v) + ? Object.fromEntries(Object.entries(v).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))) + : v, + ) + +/** Hex sha256 through Web Crypto, available on Node, Bun, Deno, and Workers. */ +export const sha256Hex = async (text: string): Promise => { + const digest = await globalThis.crypto.subtle.digest("SHA-256", new TextEncoder().encode(text)) + return Array.from(new Uint8Array(digest), (byte) => byte.toString(16).padStart(2, "0")).join("") +} diff --git a/src/schema/ops.ts b/src/schema/ops.ts new file mode 100644 index 0000000..e7a312a --- /dev/null +++ b/src/schema/ops.ts @@ -0,0 +1,135 @@ +// Migration operations: what a generated migration stores. +// +// Generated migrations are ops, not SQL text, so the cluster name and +// replicated engines are applied when the migration runs rather than frozen +// into the committed file. `renderOp` is the only place an op becomes SQL. + +import { Schema } from "effect" +import { ColumnEntity, IndexEntity, MaterializedViewEntity, TableEntity } from "./entities" +import { + ident, + renderAlter, + renderColumnDefinition, + renderCreateMaterializedView, + renderCreateTable, + renderDropTable, + renderDropView, + renderIndexDefinition, + renderSettings, + type RenderOptions, +} from "./render" + +export const MigrationOp = Schema.Union([ + Schema.Struct({ + op: Schema.Literal("create_table"), + table: TableEntity, + columns: Schema.Array(ColumnEntity), + indexes: Schema.Array(IndexEntity), + }), + Schema.Struct({ op: Schema.Literal("drop_table"), name: Schema.String }), + Schema.Struct({ op: Schema.Literal("create_view"), view: MaterializedViewEntity }), + Schema.Struct({ op: Schema.Literal("drop_view"), name: Schema.String }), + Schema.Struct({ op: Schema.Literal("add_column"), column: ColumnEntity, after: Schema.NullOr(Schema.String) }), + Schema.Struct({ op: Schema.Literal("drop_column"), table: Schema.String, name: Schema.String }), + /** Default, codec, or comment changed; the type is unchanged, so no data is rewritten. */ + Schema.Struct({ op: Schema.Literal("modify_column"), from: ColumnEntity, to: ColumnEntity }), + Schema.Struct({ op: Schema.Literal("add_index"), index: IndexEntity }), + Schema.Struct({ op: Schema.Literal("drop_index"), table: Schema.String, name: Schema.String }), + Schema.Struct({ op: Schema.Literal("modify_ttl"), table: Schema.String, ttl: Schema.NullOr(Schema.String) }), + Schema.Struct({ + op: Schema.Literal("modify_settings"), + table: Schema.String, + set: Schema.Record(Schema.String, Schema.String), + reset: Schema.Array(Schema.String), + }), + Schema.Struct({ op: Schema.Literal("modify_comment"), table: Schema.String, comment: Schema.NullOr(Schema.String) }), +]) +export type MigrationOp = typeof MigrationOp.Type + +/** The file a generated migration is written to. */ +export const MigrationFile = Schema.Struct({ + version: Schema.Literal("1"), + ops: Schema.Array(MigrationOp), +}) +export type MigrationFile = typeof MigrationFile.Type + +/** Labels the plan prints, so the expensive lines stand out. */ +export type OpLabel = "metadata" | "destructive" | "ingest gap" + +export const labelOf = (op: MigrationOp): OpLabel => { + switch (op.op) { + case "drop_table": + case "drop_column": + return "destructive" + case "drop_view": + return "ingest gap" + default: + return "metadata" + } +} + +const quote = (value: string): string => `'${value.replace(/\\/g, "\\\\").replace(/'/g, "\\'")}'` + +/** The statements one op runs, in order. Every statement is safe to repeat. */ +export const renderOp = (op: MigrationOp, options: RenderOptions = {}): ReadonlyArray => { + switch (op.op) { + case "create_table": + return [renderCreateTable(op.table, op.columns, op.indexes, options)] + case "drop_table": + return [renderDropTable(op.name, options)] + case "create_view": + return [renderCreateMaterializedView(op.view, options)] + case "drop_view": + return [renderDropView(op.name, options)] + case "add_column": + return [ + renderAlter( + op.column.table, + `ADD COLUMN IF NOT EXISTS ${renderColumnDefinition(op.column)}${op.after === null ? " FIRST" : ` AFTER ${ident(op.after)}`}`, + options, + ), + ] + case "drop_column": + // A mutation: wait for it, so the step is journaled only once the parts are rewritten. + return [`${renderAlter(op.table, `DROP COLUMN IF EXISTS ${ident(op.name)}`, options)} SETTINGS mutations_sync = 2`] + case "modify_column": { + const { from, to } = op + const name = ident(to.name) + const out: Array = [] + if (from.default !== null && to.default === null) { + out.push(renderAlter(to.table, `MODIFY COLUMN ${name} REMOVE DEFAULT`, options)) + } + if (from.codec !== null && to.codec === null) { + out.push(renderAlter(to.table, `MODIFY COLUMN ${name} REMOVE CODEC`, options)) + } + if ((to.default !== null && JSON.stringify(from.default) !== JSON.stringify(to.default)) || (to.codec !== null && from.codec !== to.codec)) { + const parts = [name, to.type] + if (to.default !== null) parts.push(`${to.default.kind} ${to.default.expr}`) + if (to.codec !== null) parts.push(`CODEC(${to.codec})`) + out.push(renderAlter(to.table, `MODIFY COLUMN ${parts.join(" ")}`, options)) + } + if (from.comment !== to.comment) { + out.push(renderAlter(to.table, `COMMENT COLUMN ${name} ${quote(to.comment ?? "")}`, options)) + } + return out + } + case "add_index": + return [renderAlter(op.index.table, `ADD INDEX IF NOT EXISTS ${renderIndexDefinition(op.index).slice("INDEX ".length)}`, options)] + case "drop_index": + return [renderAlter(op.table, `DROP INDEX IF EXISTS ${ident(op.name)}`, options)] + case "modify_ttl": + // A TTL edit must not rewrite every part of a large table as a side effect. + return [ + op.ttl === null + ? renderAlter(op.table, "REMOVE TTL", options) + : `${renderAlter(op.table, `MODIFY TTL ${op.ttl}`, options)} SETTINGS materialize_ttl_after_modify = 0`, + ] + case "modify_settings": + return [ + ...(Object.keys(op.set).length > 0 ? [renderAlter(op.table, `MODIFY SETTING ${renderSettings(op.set)}`, options)] : []), + ...op.reset.map((key) => renderAlter(op.table, `RESET SETTING ${key}`, options)), + ] + case "modify_comment": + return [renderAlter(op.table, `MODIFY COMMENT ${quote(op.comment ?? "")}`, options)] + } +} diff --git a/src/schema/render.ts b/src/schema/render.ts new file mode 100644 index 0000000..4692ad8 --- /dev/null +++ b/src/schema/render.ts @@ -0,0 +1,116 @@ +// DDL rendering from entities. +// +// Deployment shape is a render option, never part of the schema: one snapshot +// renders for a single server, a replicated cluster (`Replicated*` engines, +// `ON CLUSTER`), or ClickHouse Cloud (plain MergeTree, which Cloud converts). + +import type { + ColumnEntity, + EngineSpec, + IndexEntity, + MaterializedViewEntity, + SchemaEntity, + TableEntity, +} from "./entities" + +export interface RenderOptions { + /** Adds `ON CLUSTER ` to every statement. */ + readonly cluster?: string + /** + * Render MergeTree-family engines as `Replicated*`. The Keeper path and + * replica name default to the macros most clusters define. + */ + readonly replicated?: { readonly path?: string; readonly replica?: string } +} + +const IDENTIFIER = /^[A-Za-z_][A-Za-z0-9_]*$/ + +/** An identifier, backquoted only when it is not a plain one. */ +export const ident = (name: string): string => (IDENTIFIER.test(name) ? name : `\`${name.replace(/`/g, "``")}\``) + +const quoteString = (value: string): string => `'${value.replace(/\\/g, "\\\\").replace(/'/g, "\\'")}'` + +const onCluster = (options: RenderOptions): string => + options.cluster === undefined ? "" : ` ON CLUSTER ${ident(options.cluster)}` + +export const renderEngine = (spec: EngineSpec, options: RenderOptions = {}): string => { + if (options.replicated !== undefined && spec.family.endsWith("MergeTree")) { + const path = options.replicated.path ?? "/clickhouse/tables/{shard}/{database}/{table}" + const replica = options.replicated.replica ?? "{replica}" + return `Replicated${spec.family}(${[quoteString(path), quoteString(replica), ...spec.params].join(", ")})` + } + return spec.params.length === 0 && !spec.family.endsWith("MergeTree") + ? spec.family + : `${spec.family}(${spec.params.join(", ")})` +} + +export const renderColumnDefinition = (column: ColumnEntity): string => { + const parts = [ident(column.name), column.type] + if (column.default !== null) parts.push(`${column.default.kind} ${column.default.expr}`) + if (column.codec !== null) parts.push(`CODEC(${column.codec})`) + if (column.comment !== null) parts.push(`COMMENT ${quoteString(column.comment)}`) + return parts.join(" ") +} + +export const renderIndexDefinition = (index: IndexEntity): string => + `INDEX ${ident(index.name)} ${index.expr} TYPE ${index.type} GRANULARITY ${index.granularity}` + +export const renderSettings = (settings: Readonly>): string => + Object.entries(settings) + .map(([key, value]) => `${key} = ${value}`) + .join(", ") + +export const renderCreateTable = ( + table: TableEntity, + columns: ReadonlyArray, + indexes: ReadonlyArray, + options: RenderOptions = {}, +): string => { + const body = [ + ...[...columns].sort((a, b) => a.position - b.position).map(renderColumnDefinition), + ...indexes.map(renderIndexDefinition), + ] + const clauses = [`ENGINE = ${renderEngine(table.engine, options)}`] + if (table.partitionBy !== null) clauses.push(`PARTITION BY ${table.partitionBy}`) + if (table.primaryKey !== null) clauses.push(`PRIMARY KEY ${table.primaryKey}`) + if (table.orderBy !== null) clauses.push(`ORDER BY ${table.orderBy}`) + if (table.ttl !== null) clauses.push(`TTL ${table.ttl}`) + if (Object.keys(table.settings).length > 0) clauses.push(`SETTINGS ${renderSettings(table.settings)}`) + if (table.comment !== null) clauses.push(`COMMENT ${quoteString(table.comment)}`) + return `CREATE TABLE IF NOT EXISTS ${ident(table.name)}${onCluster(options)}\n(\n\t${body.join(",\n\t")}\n)\n${clauses.join("\n")}` +} + +export const renderCreateMaterializedView = (view: MaterializedViewEntity, options: RenderOptions = {}): string => + `CREATE MATERIALIZED VIEW IF NOT EXISTS ${ident(view.name)}${onCluster(options)} TO ${ident(view.to)}\nAS ${view.select}` + +/** `ALTER TABLE [ON CLUSTER c] `. */ +export const renderAlter = (table: string, action: string, options: RenderOptions = {}): string => + `ALTER TABLE ${ident(table)}${onCluster(options)} ${action}` + +export const renderDropTable = (name: string, options: RenderOptions = {}): string => + `DROP TABLE IF EXISTS ${ident(name)}${onCluster(options)} SYNC` + +export const renderDropView = (name: string, options: RenderOptions = {}): string => + `DROP VIEW IF EXISTS ${ident(name)}${onCluster(options)} SYNC` + +/** + * Every CREATE statement for a schema: tables (with their columns and indexes) + * first, then views, so a view's target always exists before the view. + */ +export const renderSchema = (entities: ReadonlyArray, options: RenderOptions = {}): ReadonlyArray => { + const tables = entities.filter((e): e is TableEntity => e.kind === "table") + const columns = entities.filter((e): e is ColumnEntity => e.kind === "column") + const indexes = entities.filter((e): e is IndexEntity => e.kind === "index") + const views = entities.filter((e): e is MaterializedViewEntity => e.kind === "materialized_view") + return [ + ...tables.map((table) => + renderCreateTable( + table, + columns.filter((c) => c.table === table.name), + indexes.filter((i) => i.table === table.name), + options, + ), + ), + ...views.map((view) => renderCreateMaterializedView(view, options)), + ] +} diff --git a/src/schema/schema.test.ts b/src/schema/schema.test.ts new file mode 100644 index 0000000..fcb99ec --- /dev/null +++ b/src/schema/schema.test.ts @@ -0,0 +1,192 @@ +import { Effect } from "effect" +import { describe, expect, it } from "vitest" +import * as CH from "../ch/index" +import * as S from "../schema" + +const Spans = S.defineTable("spans", { + columns: { + OrgId: CH.custom("LowCardinality(String)", CH.string.schema), + Timestamp: S.column(CH.dateTime64, { codec: "Delta, ZSTD(1)" }), + ServiceName: CH.string, + Duration: S.column(CH.uint64, { default: 0 }), + Day: S.column(CH.string, { materialized: ($) => CH.formatDateTime($.Timestamp, "%F") }), + }, + engine: S.engine.mergeTree(), + orderBy: ["OrgId", "ServiceName", "Timestamp"], + partitionBy: "toDate(Timestamp)", + ttl: S.ttlAfterDays("toDate(Timestamp)", 30), + settings: { index_granularity: 8192 }, + indexes: [S.index("idx_duration", ($) => $.Duration, "minmax")], + tenantColumn: "OrgId", +}) + +const ServiceCounts = S.defineTable("service_counts", { + columns: { OrgId: CH.string, ServiceName: CH.string, Spans: CH.uint64 }, + engine: S.engine.summingMergeTree(), + orderBy: ["OrgId", "ServiceName"], +}) + +const ServiceCountsMv = S.materializedView("service_counts_mv", { + to: ServiceCounts, + as: CH.from(Spans) + .select(($) => ({ OrgId: $.OrgId, ServiceName: $.ServiceName, Spans: CH.count() })) + .groupBy("OrgId", "ServiceName"), +}) + +describe("defineTable", () => { + it("is a Table the query builder accepts", () => { + const { sql } = CH.compileUnsafe( + CH.from(Spans).select("ServiceName").where(($) => [$.OrgId.eq("o")]), + {}, + ) + expect(sql).toContain("FROM spans") + expect(Spans.tenantColumn).toBe("OrgId") + }) + + it("renders its DDL", () => { + const [table] = S.renderSchema(S.entitiesOf([Spans])) + expect(table).toBe( + [ + "CREATE TABLE IF NOT EXISTS spans", + "(", + "\tOrgId LowCardinality(String),", + "\tTimestamp DateTime64 CODEC(Delta, ZSTD(1)),", + "\tServiceName String,", + "\tDuration UInt64 DEFAULT 0,", + "\tDay String MATERIALIZED formatDateTime(Timestamp, '%F'),", + "\tINDEX idx_duration Duration TYPE minmax GRANULARITY 1", + ")", + "ENGINE = MergeTree()", + "PARTITION BY toDate(Timestamp)", + "ORDER BY (OrgId, ServiceName, Timestamp)", + "TTL toDate(Timestamp) + toIntervalDay(30)", + "SETTINGS index_granularity = 8192", + ].join("\n"), + ) + }) + + it("renders replicated engines and ON CLUSTER as options", () => { + const [table] = S.renderSchema(S.entitiesOf([ServiceCounts]), { cluster: "main", replicated: {} }) + expect(table).toContain("CREATE TABLE IF NOT EXISTS service_counts ON CLUSTER main") + expect(table).toContain( + "ENGINE = ReplicatedSummingMergeTree('/clickhouse/tables/{shard}/{database}/{table}', '{replica}')", + ) + }) + + it("rejects a MergeTree without a sorting key", () => { + expect(() => + S.defineTable("bad", { columns: { a: CH.string }, engine: S.engine.mergeTree() }), + ).toThrow(/needs orderBy/) + }) + + it("rejects a non-identifier name", () => { + expect(() => + S.defineTable("bad name", { columns: { a: CH.string }, engine: S.engine.null() }), + ).toThrow(/plain identifier/) + }) +}) + +describe("materializedView", () => { + it("compiles its body with the DSL and records its source", () => { + expect(ServiceCountsMv.ddl.sources).toEqual(["spans"]) + const ddl = S.renderCreateMaterializedView(ServiceCountsMv.ddl) + expect(ddl).toMatch(/^CREATE MATERIALIZED VIEW IF NOT EXISTS service_counts_mv TO service_counts\nAS SELECT/) + expect(ddl).toContain("count() AS Spans") + expect(ddl).toMatch(/FROM spans\s+GROUP BY OrgId, ServiceName$/) + }) + + it("rejects at the type level an output column the target lacks", () => { + const misfit = () => + // @ts-expect-error `Nope` is not a column of service_counts + S.materializedView("bad_mv", { + to: ServiceCounts, + as: CH.from(Spans).select(($) => ({ OrgId: $.OrgId, Nope: $.ServiceName })), + }) + expect(misfit).toBeTypeOf("function") + }) + + it("rejects a view whose target is not in the schema", () => { + expect(() => S.entitiesOf([Spans, ServiceCountsMv])).toThrow(/not a table in this schema/) + }) +}) + +describe("snapshots", () => { + it("hash the entities, independent of definition order and parents", async () => { + const a = await Effect.runPromise(S.makeSnapshot(S.entitiesOf([Spans, ServiceCounts, ServiceCountsMv]), [])) + const b = await Effect.runPromise( + S.makeSnapshot(S.entitiesOf([ServiceCountsMv, ServiceCounts, Spans]), ["parent"]), + ) + expect(a.id).toBe(b.id) + expect(a.id).toMatch(/^[0-9a-f]{64}$/) + }) +}) + +describe("diffSchemas", () => { + const base = S.entitiesOf([Spans, ServiceCounts, ServiceCountsMv]) + + it("creates everything from nothing, tables before views", () => { + const { ops } = S.diffSchemas([], base) + expect(ops.map((op) => op.op)).toEqual(["create_table", "create_table", "create_view"]) + }) + + it("is empty for an unchanged schema", () => { + expect(S.diffSchemas(base, base)).toEqual({ ops: [], missingHints: [], unsupported: [] }) + }) + + it("adds a column after its neighbour and recreates a changed view", () => { + const Counts2 = S.defineTable("service_counts", { + columns: { OrgId: CH.string, Env: S.column(CH.string, { default: "" }), ServiceName: CH.string, Spans: CH.uint64 }, + engine: S.engine.summingMergeTree(), + orderBy: ["OrgId", "ServiceName"], + }) + const Mv2 = S.materializedView("service_counts_mv", { + to: Counts2, + as: CH.from(Spans) + .select(($) => ({ OrgId: $.OrgId, Env: CH.lit(""), ServiceName: $.ServiceName, Spans: CH.count() })) + .groupBy("OrgId", "Env", "ServiceName"), + }) + const { ops } = S.diffSchemas(base, S.entitiesOf([Spans, Counts2, Mv2])) + expect(ops.map((op) => op.op)).toEqual(["drop_view", "add_column", "create_view"]) + expect(S.renderOp(ops[1]!)).toEqual([ + "ALTER TABLE service_counts ADD COLUMN IF NOT EXISTS Env String DEFAULT '' AFTER OrgId", + ]) + }) + + it("asks before dropping data and orders view drops before table drops", () => { + const without = S.entitiesOf([Spans]) + const unconfirmed = S.diffSchemas(base, without) + expect(unconfirmed.missingHints).toEqual([ + { type: "confirm_data_loss", kind: "table", entity: "service_counts" }, + ]) + expect(unconfirmed.ops.map((op) => op.op)).toEqual(["drop_view"]) + + const confirmed = S.diffSchemas(base, without, unconfirmed.missingHints) + expect(confirmed.ops.map((op) => op.op)).toEqual(["drop_view", "drop_table"]) + }) + + it("reports changes ALTER cannot make", () => { + const Resorted = S.defineTable("service_counts", { + columns: { OrgId: CH.string, ServiceName: CH.string, Spans: CH.uint32 }, + engine: S.engine.summingMergeTree(), + orderBy: ["ServiceName", "OrgId"], + }) + const { unsupported } = S.diffSchemas(S.entitiesOf([ServiceCounts]), S.entitiesOf([Resorted])) + expect(unsupported.map((u) => u.message)).toEqual([ + expect.stringContaining("ORDER BY"), + expect.stringContaining("type UInt64 -> UInt32"), + ]) + }) + + it("modifies TTL without materializing it", () => { + const Shorter = S.defineTable("service_counts", { + columns: { OrgId: CH.string, ServiceName: CH.string, Spans: CH.uint64 }, + engine: S.engine.summingMergeTree(), + orderBy: ["OrgId", "ServiceName"], + ttl: "now() + toIntervalDay(1)", + }) + const { ops } = S.diffSchemas(S.entitiesOf([ServiceCounts]), S.entitiesOf([Shorter])) + expect(ops.flatMap((op) => S.renderOp(op))).toEqual([ + "ALTER TABLE service_counts MODIFY TTL now() + toIntervalDay(1) SETTINGS materialize_ttl_after_modify = 0", + ]) + }) +}) diff --git a/src/schema/snapshot.ts b/src/schema/snapshot.ts new file mode 100644 index 0000000..56209c6 --- /dev/null +++ b/src/schema/snapshot.ts @@ -0,0 +1,83 @@ +// Snapshots: the schema at one point in history, as data. + +import { Effect } from "effect" +import type { MaterializedView, SchemaTable } from "./define" +import { SchemaDefinitionDefect } from "./define" +import { + canonicalJson, + entityKey, + sha256Hex, + sortEntities, + SNAPSHOT_VERSION, + type SchemaEntity, + type Snapshot, +} from "./entities" + +/** Anything a schema module may export that `generate` collects. */ +export type SchemaObject = SchemaTable | MaterializedView + +export const isSchemaObject = (value: unknown): value is SchemaObject => + typeof value === "object" && + value !== null && + "ddl" in value && + "_tag" in value && + (value._tag === "Table" || value._tag === "MaterializedView") + +/** The entities of a set of definitions, validated as one schema. */ +export const entitiesOf = (objects: ReadonlyArray): ReadonlyArray => { + const entities: Array = [] + for (const object of objects) { + if (object._tag === "Table") entities.push(object.ddl.table, ...object.ddl.columns, ...object.ddl.indexes) + else entities.push(object.ddl) + } + const seen = new Set() + const names = new Map() + for (const entity of entities) { + const key = entityKey(entity) + if (seen.has(key)) { + throw new SchemaDefinitionDefect({ object: key, message: "defined twice" }) + } + seen.add(key) + if (entity.kind === "table" || entity.kind === "materialized_view") { + const other = names.get(entity.name) + if (other !== undefined) { + throw new SchemaDefinitionDefect({ + object: entity.name, + message: `a ${entity.kind} and a ${other} share one name`, + }) + } + names.set(entity.name, entity.kind) + } + } + // A view whose target does not exist is accepted by ClickHouse at CREATE + // time, and then every insert into its source fails with UNKNOWN_TABLE. + for (const entity of entities) { + if (entity.kind === "materialized_view" && names.get(entity.to) !== "table") { + throw new SchemaDefinitionDefect({ + object: entity.name, + message: `writes to ${entity.to}, which is not a table in this schema`, + }) + } + } + return sortEntities(entities) +} + +/** A snapshot of `entities` with the given parents. Its id is the hash of the entities alone. */ +export const makeSnapshot = ( + entities: ReadonlyArray, + prevIds: ReadonlyArray, +): Effect.Effect => + Effect.promise(() => sha256Hex(canonicalJson(sortEntities(entities)))).pipe( + Effect.map( + (id): Snapshot => ({ + version: SNAPSHOT_VERSION, + dialect: "clickhouse", + id, + prevIds: [...prevIds], + entities: sortEntities(entities), + }), + ), + ) + +/** Pretty, stable snapshot JSON for committing. */ +export const serializeSnapshot = (snapshot: Snapshot): string => `${JSON.stringify(snapshot, null, "\t")}\n` diff --git a/tests/migrate.clickhouse.test.ts b/tests/migrate.clickhouse.test.ts new file mode 100644 index 0000000..59c50fc --- /dev/null +++ b/tests/migrate.clickhouse.test.ts @@ -0,0 +1,219 @@ +import { ClickhouseClient } from "@effect/sql-clickhouse" +import { Effect, Exit, Layer } from "effect" +import { describe, expect, it } from "vitest" +import * as CH from "@maple-dev/effect-orm" +import * as Migrate from "@maple-dev/effect-orm/migrate" +import * as S from "@maple-dev/effect-orm/schema" +import { endpoint } from "./clickhouse-support" + +const user = process.env.EFFECT_ORM_CLICKHOUSE_USER ?? "default" +const password = process.env.EFFECT_ORM_CLICKHOUSE_PASSWORD ?? "" + +const client = (database: string) => ClickhouseClient.layer({ url: endpoint!, username: user, password, database }) + +/** A fresh database per test, dropped afterwards. */ +const withDatabase = (body: Effect.Effect) => + Effect.gen(function* () { + const database = `eo_migrate_${Date.now()}_${Math.floor(Math.random() * 1e6)}` + const admin = yield* ClickhouseClient.ClickhouseClient + yield* admin.asCommand(admin.unsafe(`CREATE DATABASE ${database}`)) + const scoped = Layer.provideMerge( + Layer.effect( + Migrate.MigrationDriver, + Effect.gen(function* () { + const sql = yield* ClickhouseClient.ClickhouseClient + return Migrate.fromSqlClient(sql, { command: sql.asCommand }) + }), + ), + client(database), + ) + return yield* body.pipe( + Effect.provide(scoped), + Effect.ensuring(Effect.orDie(admin.asCommand(admin.unsafe(`DROP DATABASE IF EXISTS ${database} SYNC`)))), + ) + }).pipe(Effect.provide(client("default"))) + +const Events = S.defineTable("events", { + columns: { + OrgId: CH.string, + Timestamp: S.column(CH.dateTime64, { codec: "Delta, ZSTD(1)" }), + Name: CH.string, + Count: S.column(CH.uint64, { default: 1 }), + }, + engine: S.engine.mergeTree(), + orderBy: ["OrgId", "Timestamp"], + partitionBy: "toDate(Timestamp)", + ttl: S.ttlAfterDays("toDate(Timestamp)", 30), + indexes: [S.index("idx_name", ($) => $.Name, "bloom_filter(0.01)")], +}) +const Totals = S.defineTable("totals", { + columns: { OrgId: CH.string, Name: CH.string, Count: CH.uint64 }, + engine: S.engine.summingMergeTree(), + orderBy: ["OrgId", "Name"], +}) +const TotalsMv = S.materializedView("totals_mv", { + to: Totals, + as: CH.from(Events) + .select(($) => ({ OrgId: $.OrgId, Name: $.Name, Count: CH.sum($.Count) })) + .groupBy("OrgId", "Name"), +}) + +/** Generate one migration's files from two schemas, the way `effect-orm generate` does. */ +const generated = (prevEntities: ReadonlyArray, objects: ReadonlyArray, prevIds: ReadonlyArray) => + Effect.gen(function* () { + const entities = S.entitiesOf(objects) + const { ops, missingHints, unsupported } = S.diffSchemas(prevEntities, entities) + expect(missingHints).toEqual([]) + expect(unsupported).toEqual([]) + const snapshot = yield* S.makeSnapshot(entities, prevIds) + return { + entities, + snapshot, + input: { + kind: "ops" as const, + migration: JSON.stringify({ version: "1", ops }), + snapshot: S.serializeSnapshot(snapshot), + }, + } + }) + +describe("migrate", () => { + describe.skipIf(!endpoint)("live", () => { + it("applies, verifies clean, and is a no-op the second time", async () => { + await Effect.runPromise( + withDatabase( + Effect.gen(function* () { + const first = yield* generated([], [Events, Totals, TotalsMv], [S.ORIGIN_ID]) + const migrations = yield* Migrate.fromRecord({ "20261003000000_init": first.input }) + const ran = yield* Migrate.run({ migrations, strict: true }) + expect(ran.map((m) => m.name)).toEqual(["20261003000000_init"]) + expect(yield* Migrate.verify(migrations)).toEqual({ against: "20261003000000_init", drift: [] }) + expect(yield* Migrate.run({ migrations, strict: true })).toEqual([]) + expect((yield* Migrate.status(migrations)).map((s) => s.state)).toEqual(["applied"]) + }), + ), + ) + }) + + it("applies an additive change and recreates the changed view", async () => { + await Effect.runPromise( + withDatabase( + Effect.gen(function* () { + const first = yield* generated([], [Events, Totals, TotalsMv], [S.ORIGIN_ID]) + const Totals2 = S.defineTable("totals", { + columns: { OrgId: CH.string, Name: CH.string, Count: CH.uint64, Events: S.column(CH.uint64, { default: 0 }) }, + engine: S.engine.summingMergeTree(), + orderBy: ["OrgId", "Name"], + }) + const Mv2 = S.materializedView("totals_mv", { + to: Totals2, + as: CH.from(Events) + .select(($) => ({ OrgId: $.OrgId, Name: $.Name, Count: CH.sum($.Count), Events: CH.count() })) + .groupBy("OrgId", "Name"), + }) + const second = yield* generated(first.entities, [Events, Totals2, Mv2], [first.snapshot.id]) + const migrations = yield* Migrate.fromRecord({ + "20261003000000_init": first.input, + "20261003000100_events_column": second.input, + }) + yield* Migrate.run({ migrations }) + expect((yield* Migrate.verify(migrations)).drift).toEqual([]) + const sql = yield* ClickhouseClient.ClickhouseClient + yield* sql.asCommand(sql.unsafe("INSERT INTO events (OrgId, Timestamp, Name) VALUES ('o', now64(3), 'a'), ('o', now64(3), 'a')")) + const rows = yield* sql.unsafe<{ Count: string; Events: string }>("SELECT sum(Count) AS Count, sum(Events) AS Events FROM totals") + expect(rows.map((r) => [Number(r.Count), Number(r.Events)])).toEqual([[2, 2]]) + }), + ), + ) + }) + + it("journals each statement, so a failed run resumes where it stopped", async () => { + await Effect.runPromise( + withDatabase( + Effect.gen(function* () { + const broken = yield* Migrate.fromRecord({ + "20261003000000_hand": { + kind: "sql", + migration: `CREATE TABLE a (x UInt8) ENGINE = MergeTree ORDER BY x\n${Migrate.STATEMENT_BREAKPOINT}\nTHIS IS NOT SQL`, + }, + }) + const exit = yield* Effect.exit(Migrate.run({ migrations: broken })) + expect(Exit.isFailure(exit)).toBe(true) + expect(String(exit)).toContain("MigrateStepFailed") + expect((yield* Migrate.status(broken)).map((s) => s.state)).toEqual(["partial"]) + + // Not applied yet, so fixing the file is allowed. The CREATE (which + // has no IF NOT EXISTS) must not run again. + const fixed = yield* Migrate.fromRecord({ + "20261003000000_hand": { + kind: "sql", + migration: `CREATE TABLE a (x UInt8) ENGINE = MergeTree ORDER BY x\n${Migrate.STATEMENT_BREAKPOINT}\nCREATE TABLE b (x UInt8) ENGINE = MergeTree ORDER BY x`, + }, + }) + const ran = yield* Migrate.run({ migrations: fixed }) + expect(ran).toEqual([{ name: "20261003000000_hand", steps: 2, resumedSteps: 1 }]) + }), + ), + ) + }) + + it("rejects an edited migration under strict", async () => { + await Effect.runPromise( + withDatabase( + Effect.gen(function* () { + const sql = (body: string) => + Migrate.fromRecord({ "20261003000000_hand": { kind: "sql", migration: body } }) + yield* Migrate.run({ migrations: yield* sql("CREATE TABLE a (x UInt8) ENGINE = MergeTree ORDER BY x") }) + const edited = yield* sql("CREATE TABLE a (x UInt16) ENGINE = MergeTree ORDER BY x") + expect((yield* Migrate.status(edited)).map((s) => s.state)).toEqual(["changed"]) + const exit = yield* Effect.exit(Migrate.run({ migrations: edited, strict: true })) + expect(String(exit)).toContain("MigrateHashMismatch") + expect(yield* Migrate.run({ migrations: edited })).toEqual([]) + }), + ), + ) + }) + + it("refuses to run while another run holds the lease", async () => { + await Effect.runPromise( + withDatabase( + Effect.gen(function* () { + const migrations = yield* Migrate.fromRecord({}) + yield* Migrate.run({ migrations, owner: "warmup" }) + const sql = yield* ClickhouseClient.ClickhouseClient + yield* sql.asCommand( + sql.unsafe(`INSERT INTO ${Migrate.LEDGER_TABLES.lease} (owner, expires_at) VALUES ('other', now64(3) + toIntervalMinute(5))`), + ) + const exit = yield* Effect.exit(Migrate.run({ migrations, owner: "me" })) + expect(String(exit)).toContain("MigrateLeaseHeld") + }), + ), + ) + }) + + it("reports drift against the last applied snapshot", async () => { + await Effect.runPromise( + withDatabase( + Effect.gen(function* () { + const first = yield* generated([], [Events, Totals, TotalsMv], [S.ORIGIN_ID]) + const migrations = yield* Migrate.fromRecord({ "20261003000000_init": first.input }) + yield* Migrate.run({ migrations }) + const sql = yield* ClickhouseClient.ClickhouseClient + yield* sql.asCommand(sql.unsafe("ALTER TABLE events ADD COLUMN Extra String")) + yield* sql.asCommand(sql.unsafe("ALTER TABLE totals MODIFY COLUMN Count UInt32")) + yield* sql.asCommand(sql.unsafe("DROP VIEW totals_mv")) + const { drift } = yield* Migrate.verify(migrations) + expect(drift).toEqual( + expect.arrayContaining([ + { entity: "totals_mv", problem: "missing" }, + { entity: "events.Extra", problem: "unexpected" }, + { entity: "totals.Count", problem: "type", expected: "UInt64", actual: "UInt32" }, + ]), + ) + expect(drift).toHaveLength(3) + }), + ), + ) + }) + }) +}) diff --git a/tests/package-consumer.mts b/tests/package-consumer.mts index ad1999d..5581878 100644 --- a/tests/package-consumer.mts +++ b/tests/package-consumer.mts @@ -9,6 +9,9 @@ import * as Bench from "@maple-dev/effect-orm/benchmark" import { makeHttpClient } from "@maple-dev/effect-orm/benchmark/http" import { runCli } from "@maple-dev/effect-orm/benchmark/cli" import * as SQL from "@maple-dev/effect-orm/sql" +import * as S from "@maple-dev/effect-orm/schema" +import * as Migrate from "@maple-dev/effect-orm/migrate" +import { defineConfig } from "@maple-dev/effect-orm/kit" const events = CH.table("events", { id: T.uint64, name: T.string }) const query = CH.from(events) @@ -29,6 +32,26 @@ assert.equal(SQL.compile(length(CH.lit("abc")).toFragment()), "length('abc')") const invalid = Effect.runSync(Effect.exit(CH.compile(query, {}))) assert.equal(invalid._tag, "Failure") +const managed = S.defineTable("managed", { + columns: { id: T.uint64, name: S.column(T.string, { default: "" }) }, + engine: S.engine.mergeTree(), + orderBy: ["id"], +}) +const counts = S.defineTable("counts", { columns: { name: T.string, n: T.uint64 }, engine: S.engine.summingMergeTree(), orderBy: ["name"] }) +const countsMv = S.materializedView("counts_mv", { + to: counts, + as: CH.from(managed).select(($) => ({ name: $.name, n: CH.count() })).groupBy("name"), +}) +assert.match(CH.compileUnsafe(CH.from(managed).select("name"), {}).sql, /FROM managed/) +const created = S.diffSchemas([], S.entitiesOf([managed, counts, countsMv])) +assert.deepEqual(created.ops.map((op) => op.op), ["create_table", "create_table", "create_view"]) +assert.match(created.ops.flatMap((op) => S.renderOp(op)).join("\n"), /name String DEFAULT ''/) +const loaded = await Effect.runPromise( + Migrate.fromRecord({ "20260101000000_init": { kind: "ops", migration: JSON.stringify({ version: "1", ops: created.ops }) } }), +) +assert.equal(Migrate.stepsOf(loaded[0]!).length, 3) +assert.equal(defineConfig({ schema: "./schema.ts", out: "./migrations" }).out, "./migrations") + // Compile-only negative assertions verify that published declarations retain // column checking and inferred row types. const checkTypes = () => { @@ -37,6 +60,8 @@ const checkTypes = () => { // @ts-expect-error the selected ID is a string after toString const id: number = rows[0]!.id void id + // @ts-expect-error a view output column the target table lacks must not typecheck + S.materializedView("bad_mv", { to: counts, as: CH.from(managed).select(($) => ({ missing: $.name })) }) } void checkTypes console.log("Isolated tarball imports, types, compilation and codecs passed") diff --git a/tsdown.config.ts b/tsdown.config.ts index edee4c9..ade7019 100644 --- a/tsdown.config.ts +++ b/tsdown.config.ts @@ -7,6 +7,10 @@ export default defineConfig({ types: "./src/types.ts", sql: "./src/sql/index.ts", postgres: "./src/postgres.ts", + schema: "./src/schema.ts", + migrate: "./src/migrate.ts", + kit: "./src/kit.ts", + "kit/bin": "./src/kit/bin.ts", "benchmark/index": "./src/benchmark/index.ts", "benchmark/http": "./src/benchmark/http.ts", "benchmark/cli": "./src/benchmark/cli.ts", From a01db76f055996091cc5affc16a83a481b33e9ed Mon Sep 17 00:00:00 2001 From: Makisuo Date: Sat, 3 Oct 2026 23:11:58 +0200 Subject: [PATCH 2/3] Harden migration resume, staging folders, and settings diffs - Resume skips a journaled step only when its stored SQL hash matches the statement about to run. A partial migration edited above the failed statement now fails with MigrateStepChanged instead of skipping a new statement or repeating an old one. - generate removes its staging folder when a write fails, and both migration readers skip `.tmp-` folders. - fromFileSystem skips entries that are not directories. - The diff compares records canonically, so reordered settings keys no longer produce an empty migration. --- docs/migrations.md | 4 +++- src/kit/generate.ts | 28 ++++++++++++++++------------ src/kit/kit.test.ts | 12 +++++++++++- src/migrate.ts | 2 ++ src/migrate/errors.ts | 9 ++++++++- src/migrate/ledger.ts | 4 ++-- src/migrate/run.ts | 18 +++++++++++++++--- src/migrate/source.test.ts | 22 ++++++++++++++++++++++ src/migrate/source.ts | 6 ++++++ src/schema/diff.ts | 12 ++++++------ src/schema/schema.test.ts | 8 ++++++++ tests/migrate.clickhouse.test.ts | 17 +++++++++++++++++ 12 files changed, 116 insertions(+), 26 deletions(-) create mode 100644 src/migrate/source.test.ts diff --git a/docs/migrations.md b/docs/migrations.md index 283a6c4..b3dee44 100644 --- a/docs/migrations.md +++ b/docs/migrations.md @@ -153,7 +153,9 @@ How a run behaves: `strict` a changed file fails with `MigrateHashMismatch`, otherwise it logs a warning. - **Each statement is journaled** in `_effect_orm_migration_steps` after it finishes, and the migration row in `_effect_orm_migrations` is written last. A run that fails partway leaves - the migration `partial`; the next run skips the finished statements and resumes. + the migration `partial`; the next run skips the finished statements and resumes. A finished + statement whose SQL has since changed fails the run with `MigrateStepChanged` instead of being + skipped: edit only the failed statement and those after it. - **A lease** in `_effect_orm_migration_lease` stops a second run while one is active. It is best effort: two runs starting in the same instant can both proceed. Serialize deploys if that matters. diff --git a/src/kit/generate.ts b/src/kit/generate.ts index 1689adf..24a204c 100644 --- a/src/kit/generate.ts +++ b/src/kit/generate.ts @@ -1,12 +1,12 @@ // `effect-orm generate`: diff the schema modules against the migrations folder // and write the next migration. Offline: it never connects to a database. -import { mkdir, readdir, readFile, rename, stat, writeFile } from "node:fs/promises" +import { mkdir, readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises" import { join, resolve } from "node:path" import { pathToFileURL } from "node:url" import { Effect, Schema } from "effect" import type { Layer } from "effect" -import { fromRecord, type LoadedMigration, type MigrationInput } from "../migrate/source" +import { fromRecord, isStagingName, type LoadedMigration, type MigrationInput } from "../migrate/source" import type { MigrationDriver } from "../migrate/driver" import { diffSchemas, type Hint } from "../schema/diff" import { labelOf, renderOp, type MigrationFile } from "../schema/ops" @@ -62,6 +62,7 @@ export const readMigrations = (out: string): Effect.Effect = {} const optional = (path: string) => readFile(path, "utf8").then((text): string | undefined => text, () => undefined) for (const name of names.sort()) { + if (isStagingName(name)) continue const dir = join(out, name) const isDir = yield* io(() => stat(dir).then((s) => s.isDirectory()), `Cannot stat ${dir}`) if (!isDir) continue @@ -173,16 +174,19 @@ export const generate = (config: KitConfig, cwd: string, options: GenerateOption const name = `${timestamp(options.now ?? new Date(), taken)}_${options.name ?? `${pick(ADJECTIVES)}_${pick(NOUNS)}`}` const dir = join(out, name) const staging = `${dir}.tmp-${process.pid}` - yield* io(() => mkdir(staging, { recursive: true }), `Cannot create ${staging}`) - yield* io( - () => - file === undefined - ? writeFile(join(staging, "migration.sql"), "-- Custom SQL migration. Separate statements with a line holding only:\n-- --> statement-breakpoint\n") - : writeFile(join(staging, "migration.json"), `${JSON.stringify(file, null, "\t")}\n`), - `Cannot write ${dir}`, - ) - yield* io(() => writeFile(join(staging, "snapshot.json"), serializeSnapshot(snapshot)), `Cannot write ${dir}`) - yield* io(() => rename(staging, dir), `Cannot move ${staging} to ${dir}`) + // Written aside and renamed into place, so a reader never sees half a migration. + yield* Effect.gen(function* () { + yield* io(() => mkdir(staging, { recursive: true }), `Cannot create ${staging}`) + yield* io( + () => + file === undefined + ? writeFile(join(staging, "migration.sql"), "-- Custom SQL migration. Separate statements with a line holding only:\n-- --> statement-breakpoint\n") + : writeFile(join(staging, "migration.json"), `${JSON.stringify(file, null, "\t")}\n`), + `Cannot write ${dir}`, + ) + yield* io(() => writeFile(join(staging, "snapshot.json"), serializeSnapshot(snapshot)), `Cannot write ${dir}`) + yield* io(() => rename(staging, dir), `Cannot move ${staging} to ${dir}`) + }).pipe(Effect.onError(() => Effect.promise(() => rm(staging, { recursive: true, force: true }).catch(() => undefined)))) const plan = (file?.ops ?? []).map((op) => ({ label: labelOf(op), sql: renderOp(op, config.render) })) return { written: dir, plan } satisfies GenerateResult diff --git a/src/kit/kit.test.ts b/src/kit/kit.test.ts index 8df7ff9..dcf27e6 100644 --- a/src/kit/kit.test.ts +++ b/src/kit/kit.test.ts @@ -1,4 +1,4 @@ -import { mkdtempSync, readdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs" +import { mkdirSync, mkdtempSync, readdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs" import { tmpdir } from "node:os" import { join, resolve } from "node:path" import { afterEach, beforeEach, describe, expect, it } from "vitest" @@ -81,6 +81,16 @@ describe("effect-orm CLI", () => { expect(folders()).toHaveLength(2) }) + it("skips a staging folder a crashed generate left behind", async () => { + writeFileSync(join(dir, "schema.ts"), schemaModule()) + await cli("generate", "--name", "init") + const leftover = join(dir, "migrations", "20990101000000_half.tmp-4242") + mkdirSync(leftover) + writeFileSync(join(leftover, "migration.json"), '{"version":"1","ops":[]}') + expect(await cli("check")).toBe(0) + expect(lines.at(-1)).toBe("1 migrations, ok.") + }) + it("check fails when a snapshot was edited by hand", async () => { writeFileSync(join(dir, "schema.ts"), schemaModule()) await cli("generate", "--name", "init") diff --git a/src/migrate.ts b/src/migrate.ts index 905d9ae..53c843d 100644 --- a/src/migrate.ts +++ b/src/migrate.ts @@ -11,6 +11,7 @@ export { MigrateLeaseHeld, MigrateSourceError, MigrateSqlError, + MigrateStepChanged, MigrateStepFailed, type MigrateError, } from "./migrate/errors" @@ -18,6 +19,7 @@ export { LEDGER_TABLES } from "./migrate/ledger" export { STATEMENT_BREAKPOINT, fromFileSystem, + isStagingName, fromRecord, orderMigrations, stepsOf, diff --git a/src/migrate/errors.ts b/src/migrate/errors.ts index 2df7dec..7908313 100644 --- a/src/migrate/errors.ts +++ b/src/migrate/errors.ts @@ -35,4 +35,11 @@ export class MigrateStepFailed extends Schema.TaggedError()(" cause: Schema.Defect(), }) {} -export type MigrateError = MigrateSqlError | MigrateSourceError | MigrateHashMismatch | MigrateLeaseHeld | MigrateStepFailed +/** A statement a partial migration already ran has different SQL now. */ +export class MigrateStepChanged extends Schema.TaggedError()("@maple-dev/effect-orm/MigrateStepChanged", { + migration: Schema.String, + step: Schema.String, + message: Schema.String, +}) {} + +export type MigrateError = MigrateSqlError | MigrateSourceError | MigrateHashMismatch | MigrateLeaseHeld | MigrateStepChanged | MigrateStepFailed diff --git a/src/migrate/ledger.ts b/src/migrate/ledger.ts index fdca6da..74ce67b 100644 --- a/src/migrate/ledger.ts +++ b/src/migrate/ledger.ts @@ -64,8 +64,8 @@ export const readApplied = Effect.gen(function* () { export const readDoneSteps = (name: string) => Effect.gen(function* () { const driver = yield* MigrationDriver - const rows = yield* driver.query(`SELECT step FROM ${LEDGER_TABLES.steps} FINAL WHERE name = ${q(name)}`) - return new Set(rows.map((row) => String(row.step))) + const rows = yield* driver.query(`SELECT step, sql_hash FROM ${LEDGER_TABLES.steps} FINAL WHERE name = ${q(name)}`) + return new Map(rows.map((row) => [String(row.step), String(row.sql_hash)] as const)) }) export const recordStep = (name: string, step: string, sqlHash: string) => diff --git a/src/migrate/run.ts b/src/migrate/run.ts index 0660e24..57c0fed 100644 --- a/src/migrate/run.ts +++ b/src/migrate/run.ts @@ -10,7 +10,7 @@ import { Effect, Option } from "effect" import { sha256Hex } from "../schema/entities" import type { RenderOptions } from "../schema/render" import { MigrationDriver } from "./driver" -import { MigrateHashMismatch, MigrateLeaseHeld, MigrateStepFailed, type MigrateError } from "./errors" +import { MigrateHashMismatch, MigrateLeaseHeld, MigrateStepChanged, MigrateStepFailed, type MigrateError } from "./errors" import { ensureLedger, readApplied, @@ -91,10 +91,22 @@ const applyOne = (migration: LoadedMigration, render: RenderOptions, renew: Effe const done = yield* readDoneSteps(migration.name) let resumed = 0 for (const step of steps) { - if (done.has(step.id)) { + const sqlHash = yield* Effect.promise(() => sha256Hex(step.sql)) + const doneHash = done.get(step.id) + if (doneHash === sqlHash) { resumed += 1 continue } + // A finished step whose SQL is now different: the partial migration was + // edited above the failure, or rendered with other options. Skipping it + // would leave a statement unrun; rerunning it could repeat one. + if (doneHash !== undefined) { + return yield* new MigrateStepChanged({ + migration: migration.name, + step: step.id, + message: `${migration.name} step ${step.id} already ran with different SQL. Edit only the failed statement and those after it, or keep the render options of the first run`, + }) + } yield* driver.execute(step.sql).pipe( Effect.mapError( (cause) => @@ -108,7 +120,7 @@ const applyOne = (migration: LoadedMigration, render: RenderOptions, renew: Effe ), Effect.withSpan("effect_orm.migrate.step", { attributes: { "effect_orm.migration.step": step.id } }), ) - yield* recordStep(migration.name, step.id, yield* Effect.promise(() => sha256Hex(step.sql))) + yield* recordStep(migration.name, step.id, sqlHash) yield* renew } yield* recordMigration(migration.name, migration.hash) diff --git a/src/migrate/source.test.ts b/src/migrate/source.test.ts new file mode 100644 index 0000000..1ec5617 --- /dev/null +++ b/src/migrate/source.test.ts @@ -0,0 +1,22 @@ +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { BunServices } from "@effect/platform-bun" +import { Effect } from "effect" +import { afterEach, describe, expect, it } from "vitest" +import { fromFileSystem } from "./source" + +describe("fromFileSystem", () => { + const dir = mkdtempSync(join(tmpdir(), "effect-orm-source-")) + afterEach(() => rmSync(dir, { recursive: true, force: true })) + + it("reads migration folders and skips files and staging folders beside them", async () => { + mkdirSync(join(dir, "20260101000000_init")) + writeFileSync(join(dir, "20260101000000_init", "migration.sql"), "SELECT 1\n--> statement-breakpoint\nSELECT 2") + mkdirSync(join(dir, "20260102000000_half.tmp-123")) + writeFileSync(join(dir, "20260102000000_half.tmp-123", "migration.sql"), "SELECT 3") + writeFileSync(join(dir, "README.md"), "notes") + const migrations = await Effect.runPromise(fromFileSystem(dir).pipe(Effect.provide(BunServices.layer))) + expect(migrations.map((m) => [m.name, m.sql])).toEqual([["20260101000000_init", ["SELECT 1", "SELECT 2"]]]) + }) +}) diff --git a/src/migrate/source.ts b/src/migrate/source.ts index e840989..02b229c 100644 --- a/src/migrate/source.ts +++ b/src/migrate/source.ts @@ -13,6 +13,9 @@ import { MigrateSourceError } from "./errors" export const STATEMENT_BREAKPOINT = "--> statement-breakpoint" +/** `generate` writes a migration into `.tmp-` and renames it into place. Readers skip these. */ +export const isStagingName = (name: string): boolean => /\.tmp-\d+$/.test(name) + export interface MigrationInput { /** `migration.json` or `migration.sql` contents. */ readonly migration: string @@ -144,7 +147,10 @@ export const fromFileSystem = ( const entries = yield* fs.readDirectory(directory).pipe(Effect.mapError(fail(directory, "cannot read the migrations directory"))) const record: Record = {} for (const name of entries) { + if (isStagingName(name)) continue const dir = path.join(directory, name) + const info = yield* fs.stat(dir).pipe(Effect.mapError(fail(name, "cannot stat"))) + if (info.type !== "Directory") continue const read = (file: string) => fs.exists(path.join(dir, file)).pipe( Effect.flatMap((exists) => (exists ? Effect.map(fs.readFileString(path.join(dir, file)), (text): string | undefined => text) : Effect.succeed(undefined))), diff --git a/src/schema/diff.ts b/src/schema/diff.ts index f1799c1..6de748e 100644 --- a/src/schema/diff.ts +++ b/src/schema/diff.ts @@ -15,6 +15,7 @@ import { Schema } from "effect" import { + canonicalJson, entityKey, type ColumnEntity, type IndexEntity, @@ -56,7 +57,8 @@ const byKind = (entities: ReadonlyArray [entityKey(e), e] as const), ) -const same = (a: unknown, b: unknown): boolean => JSON.stringify(a) === JSON.stringify(b) +/** Key order is not a change; the snapshot hash ignores it too. */ +const same = (a: unknown, b: unknown): boolean => canonicalJson(a) === canonicalJson(b) const opOrder: Record = { drop_view: 0, @@ -202,11 +204,9 @@ const diffTable = ( if (before.partitionBy !== after.partitionBy) rebuild("PARTITION BY", before.partitionBy, after.partitionBy) if (before.primaryKey !== after.primaryKey) rebuild("PRIMARY KEY", before.primaryKey, after.primaryKey) if (before.ttl !== after.ttl) ops.push({ op: "modify_ttl", table: after.name, ttl: after.ttl }) - if (!same(before.settings, after.settings)) { - const set = Object.fromEntries(Object.entries(after.settings).filter(([k, v]) => before.settings[k] !== v)) - const reset = Object.keys(before.settings).filter((k) => !(k in after.settings)) - ops.push({ op: "modify_settings", table: after.name, set, reset }) - } + const set = Object.fromEntries(Object.entries(after.settings).filter(([k, v]) => before.settings[k] !== v)) + const reset = Object.keys(before.settings).filter((k) => !(k in after.settings)) + if (Object.keys(set).length > 0 || reset.length > 0) ops.push({ op: "modify_settings", table: after.name, set, reset }) if (before.comment !== after.comment) ops.push({ op: "modify_comment", table: after.name, comment: after.comment }) } diff --git a/src/schema/schema.test.ts b/src/schema/schema.test.ts index fcb99ec..7953b03 100644 --- a/src/schema/schema.test.ts +++ b/src/schema/schema.test.ts @@ -189,4 +189,12 @@ describe("diffSchemas", () => { "ALTER TABLE service_counts MODIFY TTL now() + toIntervalDay(1) SETTINGS materialize_ttl_after_modify = 0", ]) }) + + it("ignores settings that only changed key order", () => { + const make = (settings: Record) => + S.defineTable("ordered", { columns: { a: CH.string }, engine: S.engine.mergeTree(), orderBy: ["a"], settings }) + const before = S.entitiesOf([make({ index_granularity: 8192, merge_with_ttl_timeout: 3600 })]) + const after = S.entitiesOf([make({ merge_with_ttl_timeout: 3600, index_granularity: 8192 })]) + expect(S.diffSchemas(before, after).ops).toEqual([]) + }) }) diff --git a/tests/migrate.clickhouse.test.ts b/tests/migrate.clickhouse.test.ts index 59c50fc..b7b85d8 100644 --- a/tests/migrate.clickhouse.test.ts +++ b/tests/migrate.clickhouse.test.ts @@ -157,6 +157,23 @@ describe("migrate", () => { ) }) + it("refuses to resume a partial migration edited above the failed statement", async () => { + await Effect.runPromise( + withDatabase( + Effect.gen(function* () { + const source = (...statements: Array) => + Migrate.fromRecord({ "20261003000000_hand": { kind: "sql", migration: statements.join(`\n${Migrate.STATEMENT_BREAKPOINT}\n`) } }) + const createA = "CREATE TABLE a (x UInt8) ENGINE = MergeTree ORDER BY x" + yield* Effect.exit(Migrate.run({ migrations: yield* source(createA, "THIS IS NOT SQL") })) + const inserted = yield* source("CREATE TABLE z (x UInt8) ENGINE = MergeTree ORDER BY x", createA, "SELECT 1") + const exit = yield* Effect.exit(Migrate.run({ migrations: inserted })) + expect(String(exit)).toContain("MigrateStepChanged") + expect((yield* Migrate.status(inserted)).map((s) => s.state)).toEqual(["partial"]) + }), + ), + ) + }) + it("rejects an edited migration under strict", async () => { await Effect.runPromise( withDatabase( From 02f0be0c2ed69485c149e04785343afeab64fc54 Mon Sep 17 00:00:00 2001 From: Makisuo Date: Sat, 3 Oct 2026 23:20:10 +0200 Subject: [PATCH 3/3] Never re-run a migration statement whose outcome is unknown The step journal now records `started` before a statement runs, then `done` or `failed`. If the statement succeeded but its `done` row was never written (or the process died), the step stays `started`: the next run stops there with MigrateStepUncertain and status reports `uncertain`, instead of executing it again. After checking the database, `resolveStep` (`effect-orm resolve --ran | --not-ran`) records what happened. Statements the server rejected are journaled `failed` and run again as before. Journal rows carry an explicit, strictly increasing version so the latest state wins even within one millisecond. --- docs/migrations.md | 5 +++ src/kit/cli.ts | 23 ++++++++++++ src/migrate.ts | 12 +++++- src/migrate/errors.ts | 11 +++++- src/migrate/ledger.ts | 41 +++++++++++++++++---- src/migrate/run.ts | 63 +++++++++++++++++++++++++++----- tests/migrate.clickhouse.test.ts | 37 +++++++++++++++++++ 7 files changed, 172 insertions(+), 20 deletions(-) diff --git a/docs/migrations.md b/docs/migrations.md index b3dee44..b56cdfd 100644 --- a/docs/migrations.md +++ b/docs/migrations.md @@ -156,6 +156,11 @@ How a run behaves: the migration `partial`; the next run skips the finished statements and resumes. A finished statement whose SQL has since changed fails the run with `MigrateStepChanged` instead of being skipped: edit only the failed statement and those after it. +- **Uncertain statements are never repeated.** `started` is journaled before a statement runs. If + the process dies, or the `done` row cannot be written, the statement may or may not have run, and + the next run stops there with `MigrateStepUncertain` (status `uncertain`). Check the database, + then record what happened: `Migrate.resolveStep` or `effect-orm resolve + --ran | --not-ran`. A statement the server rejected is journaled `failed` and simply runs again. - **A lease** in `_effect_orm_migration_lease` stops a second run while one is active. It is best effort: two runs starting in the same instant can both proceed. Serialize deploys if that matters. diff --git a/src/kit/cli.ts b/src/kit/cli.ts index 91657f0..d3b9937 100644 --- a/src/kit/cli.ts +++ b/src/kit/cli.ts @@ -16,6 +16,8 @@ const help = `effect-orm: schema migrations for ClickHouse migrate [--strict] Apply pending migrations (needs config.driver) status Applied, pending, partial, or changed, per migration verify Compare the database with the last applied snapshot + resolve Record what a step left "started" did, after checking the database: + --ran | --not-ran --ran skips it on the next migrate, --not-ran runs it --config Default: effect-orm.config.ts --json Machine-readable output @@ -29,6 +31,8 @@ const options = { hints: { type: "string" }, "hints-file": { type: "string" }, strict: { type: "boolean" }, + ran: { type: "boolean" }, + "not-ran": { type: "boolean" }, json: { type: "boolean" }, help: { type: "boolean" }, } as const @@ -108,6 +112,25 @@ const program = (args: ReadonlyArray, print: (line: string) => void) => ]) return 0 } + case "resolve": { + const [, migrationName, stepId] = parsed.positionals + const ran = parsed.values.ran === true + const notRan = parsed.values["not-ran"] === true + if (migrationName === undefined || stepId === undefined || ran === notRan) { + return yield* new KitError({ code: "config", message: "usage: effect-orm resolve --ran | --not-ran" }) + } + const migrations = yield* readMigrations(resolve(cwd, config.out)) + const migration = migrations.find((m) => m.name === migrationName) + if (migration === undefined) return yield* new KitError({ code: "config", message: `no migration named ${migrationName}` }) + yield* withDriver( + config, + Migrate.resolveStep({ migration, step: stepId, outcome: ran ? "ran" : "not-ran", render: config.render ?? {} }), + ) + out({ migration: migrationName, step: stepId, outcome: ran ? "ran" : "not-ran" }, () => [ + `Recorded ${migrationName} step ${stepId} as ${ran ? "done; migrate will skip it" : "not run; migrate will run it"}.`, + ]) + return 0 + } case "migrate": case "status": case "verify": { diff --git a/src/migrate.ts b/src/migrate.ts index 53c843d..b3e066e 100644 --- a/src/migrate.ts +++ b/src/migrate.ts @@ -13,6 +13,7 @@ export { MigrateSqlError, MigrateStepChanged, MigrateStepFailed, + MigrateStepUncertain, type MigrateError, } from "./migrate/errors" export { LEDGER_TABLES } from "./migrate/ledger" @@ -27,5 +28,14 @@ export { type MigrationInput, type MigrationStep, } from "./migrate/source" -export { run, status, type AppliedMigration, type MigrationState, type MigrationStatus, type RunOptions } from "./migrate/run" +export { + resolveStep, + run, + status, + type AppliedMigration, + type MigrationState, + type MigrationStatus, + type ResolveStepOptions, + type RunOptions, +} from "./migrate/run" export { verify, type Drift, type VerifyResult } from "./migrate/verify" diff --git a/src/migrate/errors.ts b/src/migrate/errors.ts index 7908313..e378843 100644 --- a/src/migrate/errors.ts +++ b/src/migrate/errors.ts @@ -42,4 +42,13 @@ export class MigrateStepChanged extends Schema.TaggedError() message: Schema.String, }) {} -export type MigrateError = MigrateSqlError | MigrateSourceError | MigrateHashMismatch | MigrateLeaseHeld | MigrateStepChanged | MigrateStepFailed +/** + * A statement started and never reported back: it may or may not have run. + * Check the database, then record what happened with `resolveStep`. + */ +export class MigrateStepUncertain extends Schema.TaggedError()( + "@maple-dev/effect-orm/MigrateStepUncertain", + { migration: Schema.String, step: Schema.String, sql: Schema.String, message: Schema.String }, +) {} + +export type MigrateError = MigrateSqlError | MigrateSourceError | MigrateHashMismatch | MigrateLeaseHeld | MigrateStepChanged | MigrateStepFailed | MigrateStepUncertain diff --git a/src/migrate/ledger.ts b/src/migrate/ledger.ts index 74ce67b..81a01fb 100644 --- a/src/migrate/ledger.ts +++ b/src/migrate/ledger.ts @@ -5,8 +5,11 @@ // // - `_effect_orm_migrations`: one row per applied migration, written only // after every statement in it finished. Never written first. -// - `_effect_orm_migration_steps`: one row per finished statement, so a run -// that stopped halfway resumes at the first statement without a row. +// - `_effect_orm_migration_steps`: the latest state of each statement: +// `started` is written before it runs, then `done` or `failed`. A run that +// stopped halfway resumes after the last `done`. A statement left at +// `started` may or may not have run (the process died, or the `done` row was +// never written), so it is never re-run automatically; see `resolveStep`. // - `_effect_orm_migration_lease`: who is migrating, until when. Best effort: // two runs that start within the same instant can both see an empty lease. // Callers that need a hard guarantee serialize runs themselves. @@ -40,7 +43,7 @@ export const ensureLedger = (options: RenderOptions): Effect.Effect ({ name: String(row.name), hash: String(row.hash), appliedAt: String(row.applied) })) }) -export const readDoneSteps = (name: string) => +export type StepState = "started" | "done" | "failed" + +export interface StepRow { + readonly sqlHash: string + readonly state: StepState +} + +/** The latest state of each statement of a migration. */ +export const readSteps = (name: string) => Effect.gen(function* () { const driver = yield* MigrationDriver - const rows = yield* driver.query(`SELECT step, sql_hash FROM ${LEDGER_TABLES.steps} FINAL WHERE name = ${q(name)}`) - return new Map(rows.map((row) => [String(row.step), String(row.sql_hash)] as const)) + const rows = yield* driver.query(`SELECT step, sql_hash, state FROM ${LEDGER_TABLES.steps} FINAL WHERE name = ${q(name)}`) + return new Map( + rows.map((row) => { + const state = String(row.state) + const parsed: StepState = state === "started" || state === "failed" ? state : "done" + return [String(row.step), { sqlHash: String(row.sql_hash), state: parsed }] as const + }), + ) }) -export const recordStep = (name: string, step: string, sqlHash: string) => +// The ReplacingMergeTree version: strictly increasing within a process, so a +// state written after another always wins even inside one millisecond. +let lastSeq = 0 +const nextSeq = (): number => { + lastSeq = Math.max(lastSeq + 1, Date.now() * 1000) + return lastSeq +} + +export const recordStep = (name: string, step: string, sqlHash: string, state: StepState) => Effect.gen(function* () { const driver = yield* MigrationDriver yield* driver.execute( - `INSERT INTO ${LEDGER_TABLES.steps} (name, step, sql_hash) VALUES (${q(name)}, ${q(step)}, ${q(sqlHash)})`, + `INSERT INTO ${LEDGER_TABLES.steps} (name, step, sql_hash, state, seq) VALUES (${q(name)}, ${q(step)}, ${q(sqlHash)}, ${q(state)}, ${nextSeq()})`, ) }) diff --git a/src/migrate/run.ts b/src/migrate/run.ts index 57c0fed..8104e62 100644 --- a/src/migrate/run.ts +++ b/src/migrate/run.ts @@ -10,11 +10,11 @@ import { Effect, Option } from "effect" import { sha256Hex } from "../schema/entities" import type { RenderOptions } from "../schema/render" import { MigrationDriver } from "./driver" -import { MigrateHashMismatch, MigrateLeaseHeld, MigrateStepChanged, MigrateStepFailed, type MigrateError } from "./errors" +import { MigrateHashMismatch, MigrateLeaseHeld, MigrateSourceError, MigrateStepChanged, MigrateStepFailed, MigrateStepUncertain, type MigrateError } from "./errors" import { ensureLedger, readApplied, - readDoneSteps, + readSteps, readLiveLeases, recordMigration, recordStep, @@ -41,7 +41,8 @@ export interface AppliedMigration { readonly resumedSteps: number } -export type MigrationState = "applied" | "pending" | "partial" | "changed" +/** `uncertain`: a statement started and never reported back; see `resolveStep`. */ +export type MigrationState = "applied" | "pending" | "partial" | "uncertain" | "changed" export interface MigrationStatus { readonly name: string @@ -88,26 +89,38 @@ const applyOne = (migration: LoadedMigration, render: RenderOptions, renew: Effe Effect.gen(function* () { const driver = yield* MigrationDriver const steps = stepsOf(migration, render) - const done = yield* readDoneSteps(migration.name) + const journal = yield* readSteps(migration.name) let resumed = 0 for (const step of steps) { const sqlHash = yield* Effect.promise(() => sha256Hex(step.sql)) - const doneHash = done.get(step.id) - if (doneHash === sqlHash) { + const previous = journal.get(step.id) + if (previous?.state === "started") { + return yield* new MigrateStepUncertain({ + migration: migration.name, + step: step.id, + sql: step.sql, + message: `${migration.name} step ${step.id} started and never reported back, so it may or may not have run. Check the database, then record the outcome with resolveStep (effect-orm resolve)`, + }) + } + if (previous?.state === "done" && previous.sqlHash === sqlHash) { resumed += 1 continue } // A finished step whose SQL is now different: the partial migration was // edited above the failure, or rendered with other options. Skipping it // would leave a statement unrun; rerunning it could repeat one. - if (doneHash !== undefined) { + if (previous?.state === "done") { return yield* new MigrateStepChanged({ migration: migration.name, step: step.id, message: `${migration.name} step ${step.id} already ran with different SQL. Edit only the failed statement and those after it, or keep the render options of the first run`, }) } + // Journal the attempt first: if the statement runs but the `done` row is + // never written, the next run stops at this step instead of repeating it. + yield* recordStep(migration.name, step.id, sqlHash, "started") yield* driver.execute(step.sql).pipe( + Effect.tapError(() => recordStep(migration.name, step.id, sqlHash, "failed").pipe(Effect.ignore)), Effect.mapError( (cause) => new MigrateStepFailed({ @@ -120,7 +133,7 @@ const applyOne = (migration: LoadedMigration, render: RenderOptions, renew: Effe ), Effect.withSpan("effect_orm.migrate.step", { attributes: { "effect_orm.migration.step": step.id } }), ) - yield* recordStep(migration.name, step.id, sqlHash) + yield* recordStep(migration.name, step.id, sqlHash, "done") yield* renew } yield* recordMigration(migration.name, migration.hash) @@ -172,9 +185,39 @@ export const status = ( const state: MigrationState = row.value.hash === migration.hash ? "applied" : "changed" return { name: migration.name, state, appliedAt: row.value.appliedAt } } - const done = yield* readDoneSteps(migration.name) - const state: MigrationState = done.size > 0 ? "partial" : "pending" + const journal = [...(yield* readSteps(migration.name)).values()] + const state: MigrationState = journal.some((step) => step.state === "started") + ? "uncertain" + : journal.some((step) => step.state === "done") + ? "partial" + : "pending" return { name: migration.name, state, appliedAt: undefined } }), ) }) + +export interface ResolveStepOptions { + readonly migration: LoadedMigration + /** The step id from `MigrateStepUncertain`. */ + readonly step: string + /** What checking the database showed: the statement took effect, or it did not. */ + readonly outcome: "ran" | "not-ran" + readonly render?: RenderOptions +} + +/** + * Record the outcome of a step left uncertain, after checking the database. + * `ran` marks it done so the next run skips it; `not-ran` marks it failed so + * the next run executes it. + */ +export const resolveStep = (options: ResolveStepOptions): Effect.Effect => + Effect.gen(function* () { + const { migration } = options + const step = stepsOf(migration, options.render ?? {}).find((s) => s.id === options.step) + if (step === undefined) { + return yield* new MigrateSourceError({ migration: migration.name, message: `has no step ${options.step}` }) + } + yield* ensureLedger(options.render ?? {}) + const sqlHash = yield* Effect.promise(() => sha256Hex(step.sql)) + yield* recordStep(migration.name, step.id, sqlHash, options.outcome === "ran" ? "done" : "failed") + }) diff --git a/tests/migrate.clickhouse.test.ts b/tests/migrate.clickhouse.test.ts index b7b85d8..4e558c1 100644 --- a/tests/migrate.clickhouse.test.ts +++ b/tests/migrate.clickhouse.test.ts @@ -174,6 +174,43 @@ describe("migrate", () => { ) }) + it("stops at a statement that started and never reported back, until it is resolved", async () => { + await Effect.runPromise( + withDatabase( + Effect.gen(function* () { + const migrations = yield* Migrate.fromRecord({ + "20261003000000_hand": { + kind: "sql", + migration: `CREATE TABLE counts (n UInt8) ENGINE = MergeTree ORDER BY n\n${Migrate.STATEMENT_BREAKPOINT}\nINSERT INTO counts VALUES (1)`, + }, + }) + const sql = yield* ClickhouseClient.ClickhouseClient + // The INSERT ran but its `done` row was never written: what a crash between the two leaves. + yield* sql.asCommand(sql.unsafe("CREATE TABLE counts (n UInt8) ENGINE = MergeTree ORDER BY n")) + yield* Migrate.run({ migrations: yield* Migrate.fromRecord({}) }) + const [migration] = migrations + const steps = Migrate.stepsOf(migration!) + yield* Migrate.resolveStep({ migration: migration!, step: "0", outcome: "ran" }) + yield* sql.asCommand(sql.unsafe("INSERT INTO counts VALUES (1)")) + yield* sql.asCommand( + sql.unsafe( + `INSERT INTO ${Migrate.LEDGER_TABLES.steps} (name, step, sql_hash, state, seq) VALUES ('20261003000000_hand', '1', 'x', 'started', ${Date.now() * 1000 + 999})`, + ), + ) + expect(steps).toHaveLength(2) + expect((yield* Migrate.status(migrations)).map((s) => s.state)).toEqual(["uncertain"]) + const exit = yield* Effect.exit(Migrate.run({ migrations })) + expect(String(exit)).toContain("MigrateStepUncertain") + + yield* Migrate.resolveStep({ migration: migration!, step: "1", outcome: "ran" }) + expect(yield* Migrate.run({ migrations })).toEqual([{ name: "20261003000000_hand", steps: 2, resumedSteps: 2 }]) + const rows = yield* sql.unsafe<{ c: string }>("SELECT count() AS c FROM counts") + expect(Number(rows[0]?.c)).toBe(1) + }), + ), + ) + }) + it("rejects an edited migration under strict", async () => { await Effect.runPromise( withDatabase(