diff --git a/.github/workflows/publish-nuget.yml b/.github/workflows/publish-nuget.yml index 2227b0f..8714bcf 100644 --- a/.github/workflows/publish-nuget.yml +++ b/.github/workflows/publish-nuget.yml @@ -25,6 +25,9 @@ env: jobs: publish: runs-on: ubuntu-latest + outputs: + VERSION: ${{ steps.get_version.outputs.VERSION }} + CONFIGURATION: ${{ steps.get_version.outputs.CONFIGURATION }} steps: - name: Checkout code @@ -187,6 +190,26 @@ jobs: --output ./artifacts \ /p:PackageVersion=${{ steps.get_version.outputs.VERSION }} + dotnet pack src/MicroPlumberd.Migration/MicroPlumberd.Migration.csproj \ + --configuration ${{ steps.get_version.outputs.CONFIGURATION }} \ + --no-build \ + --output ./artifacts \ + /p:PackageVersion=${{ steps.get_version.outputs.VERSION }} + + dotnet pack src/MicroPlumberd.Migration.Scripting/MicroPlumberd.Migration.Scripting.csproj \ + --configuration ${{ steps.get_version.outputs.CONFIGURATION }} \ + --no-build \ + --output ./artifacts \ + /p:PackageVersion=${{ steps.get_version.outputs.VERSION }} + + # mp-rewrite ships as a dotnet tool. The package BUNDLES its dependencies (Jint, the migration + # libraries) under tools/net10.0/any, so it installs without them being on the feed first. + dotnet pack src/MicroPlumberd.Rewrite/MicroPlumberd.Rewrite.csproj \ + --configuration ${{ steps.get_version.outputs.CONFIGURATION }} \ + --no-build \ + --output ./artifacts \ + /p:PackageVersion=${{ steps.get_version.outputs.VERSION }} + - name: List artifacts run: ls -lh ./artifacts @@ -211,3 +234,47 @@ jobs: name: nuget-packages-${{ steps.get_version.outputs.VERSION }} path: ./artifacts/*.nupkg retention-days: 30 + + # Neurons have no SDK, so the tool also ships as one self-contained file per architecture, attached to the + # release for the tag. Runs only for a tag push: a workflow_dispatch has no tag to release against. + rewrite-binaries: + needs: publish + if: startsWith(github.ref, 'refs/tags/') + runs-on: ubuntu-latest + permissions: + contents: write + + steps: + - name: Checkout code + uses: actions/checkout@v4 + + - name: Setup .NET + uses: actions/setup-dotnet@v4 + with: + dotnet-version: '10.0.x' + + - name: Publish single-file binaries + run: | + for RID in linux-x64 linux-arm64; do + # Always Release: a preview TAG is about the package version, not about shipping an unoptimised + # binary to a neuron that has no SDK to rebuild it with. + dotnet publish src/MicroPlumberd.Rewrite/MicroPlumberd.Rewrite.csproj \ + --configuration Release \ + --runtime "$RID" \ + --self-contained true \ + -p:PublishSingleFile=true \ + -p:IncludeNativeLibrariesForSelfExtract=true \ + -p:EnableCompressionInSingleFile=true \ + -p:Version=${{ needs.publish.outputs.VERSION }} \ + --output "./binaries/$RID" + # One file is the whole point — an operator scps it onto a neuron and runs it. + mv "./binaries/$RID/mp-rewrite" "./binaries/mp-rewrite-${{ needs.publish.outputs.VERSION }}-$RID" + done + ls -lh ./binaries/mp-rewrite-* + + - name: Attach binaries to the release + uses: softprops/action-gh-release@v2 + with: + tag_name: ${{ github.ref_name }} + files: ./binaries/mp-rewrite-* + fail_on_unmatched_files: true diff --git a/.github/workflows/rewrite-e2e.yml b/.github/workflows/rewrite-e2e.yml new file mode 100644 index 0000000..0cd7d67 --- /dev/null +++ b/.github/workflows/rewrite-e2e.yml @@ -0,0 +1,103 @@ +# End-to-end suite for mp-rewrite (epic-095). +# +# These tests are the acceptance bar for the tool: they start REAL KurrentDB containers, bind-mount a data +# directory, rewrite it, swap it in and read the result back through the original container. They cannot be +# replaced by unit tests, because the whole value of the tool is the orchestration. +name: mp-rewrite end-to-end + +on: + pull_request: + branches: [master] + push: + branches: [master, 'feature/**'] + workflow_dispatch: + +# One run per ref. These scenarios start real containers on a shared docker host and the tool refuses when a +# sibling of the same compose project is up, so two concurrent runs would refuse each other. +concurrency: + group: rewrite-e2e-${{ github.ref }} + cancel-in-progress: true + +env: + DOTNET_VERSION: '10.0.x' + DOTNET_NOLOGO: 'true' + DOTNET_CLI_TELEMETRY_OPTOUT: 'true' + +jobs: + e2e: + runs-on: ubuntu-latest + # The suite starts two containers per scenario and runs strictly sequentially; ~7 minutes on saturn. + # 30 minutes leaves room for a cold image pull and a slower runner without letting a hang burn an hour. + timeout-minutes: 30 + + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Setup .NET + uses: actions/setup-dotnet@v4 + with: + dotnet-version: ${{ env.DOTNET_VERSION }} + + - name: Restore + run: dotnet restore src/MicroPlumberd.Rewrite.Tests/MicroPlumberd.Rewrite.Tests.csproj + + - name: Build + run: >- + dotnet build src/MicroPlumberd.Rewrite.Tests/MicroPlumberd.Rewrite.Tests.csproj + --configuration Release --no-restore + + # The fixture pulls this itself if it is missing, but doing it here gives the pull its own step, its own + # log and its own failure — rather than surfacing as a container-start timeout inside the first scenario. + - name: Pull KurrentDB image + run: docker pull docker.kurrent.io/kurrent-latest/kurrentdb:latest + + # Sequential is not optional: two fixtures up at once are siblings of each other's compose project and + # the tool would correctly refuse. xunit.runner.json pins it too, so a local run behaves the same. + - name: Run end-to-end suite + run: >- + dotnet test src/MicroPlumberd.Rewrite.Tests/MicroPlumberd.Rewrite.Tests.csproj + --configuration Release --no-build + --logger "trx;LogFileName=rewrite-e2e.trx" + --results-directory ${{ github.workspace }}/test-results + -- xUnit.ParallelizeTestCollections=false + + # A discovery break runs ZERO tests and dotnet test exits 0 — green, with nothing to notice it in. For a + # suite that is the acceptance bar for a tool whose failure mode is losing a production store, "no tests + # ran" must be a failure, and the floor must be a number someone has to consciously lower. + - name: Require the suite to have actually run + if: always() + run: | + TRX=$(ls ${{ github.workspace }}/test-results/*.trx 2>/dev/null | head -1) + if [ -z "$TRX" ]; then echo "::error::No TRX produced — the suite did not run."; exit 1; fi + EXECUTED=$(grep -o 'executed="[0-9]*"' "$TRX" | head -1 | grep -o '[0-9]*') + PASSED=$(grep -o 'passed="[0-9]*"' "$TRX" | head -1 | grep -o '[0-9]*') + OUTCOME=$(grep -o 'outcome="[A-Za-z]*"' "$TRX" | tail -1 | sed 's/.*="\(.*\)"/\1/') + echo "executed=$EXECUTED passed=$PASSED outcome=$OUTCOME" + # Measured on saturn 2026-09-07: a run that ABORTED after 71 of 76 tests still printed "Passed!" + # for the 71 that had run. A dying run looks green, so the count is the only thing that catches it — + # and a floor far below the real total would have let that very run through. Raise this + # deliberately when tests are added; never lower it to make a red build green. + MIN_TESTS=83 + if [ -z "$EXECUTED" ] || [ "$EXECUTED" -lt "$MIN_TESTS" ]; then + echo "::error::Only ${EXECUTED:-0} test(s) ran; expected at least $MIN_TESTS. A run that dies partway still reports success." + exit 1 + fi + if [ "$OUTCOME" != "Completed" ]; then + echo "::error::Test run outcome was '$OUTCOME', not 'Completed'." + exit 1 + fi + + - name: Upload test results + if: always() + uses: actions/upload-artifact@v4 + with: + name: rewrite-e2e-trx + path: ${{ github.workspace }}/test-results/*.trx + retention-days: 14 + + # A scenario that fails midway can leave a container behind; the runner is thrown away, but leaving them + # named in the log makes a CI-only failure diagnosable. Only ever by the exact name prefix this suite owns. + - name: Report leftover containers + if: always() + run: docker ps -a --filter "name=mp-rewrite-" --format '{{.Names}}\t{{.Status}}' || true diff --git a/dev-log-index-backed-merge.md b/dev-log-index-backed-merge.md new file mode 100644 index 0000000..cac581d --- /dev/null +++ b/dev-log-index-backed-merge.md @@ -0,0 +1,129 @@ +# Dev-log: Index-backed merge streams (v1, shape-A) — implementation + +Branch: `feature/mp-index-backed-merge`. Implements v1 (slices S1–S4) of +`docs/design-index-backed-merge-streams.md` exactly. Shape-B (S5) is out of scope (v2). No git actions performed +by the engineer — reviewed branch only. + +Build with `dotnet.exe`. Every slice ends with a green integration test against a REAL KurrentDB 26.1 +(`MicroPlumberd.Testing` `EventStoreServer.StartInDocker`, NEVER Testcontainers, no `ClearAllPools`). + +## What shipped, by slice + +### S1 — index-backed subscription primitive + shared subscription seam (core) +- **Relocated the index primitives into core** as `MicroPlumberd.UserDefinedIndex` and made `KurrentHttpEndpoint` + a public core utility (added `FromSettings(KurrentDBClientSettings)` alongside the existing + `Parse(connectionString)`). `MicroPlumberd.Migration` now has a `ProjectReference` → core and its + `UserDefinedIndexSource` DELEGATES create/name/filter/stream to the core type — ONE implementation + (CLAUDE.md "NEVER DUPLICATE"), dependency arrow Migration → core (design "Assembly layering", team-lead + APPROVED). `UserDefinedIndexSource` keeps its offline-only members (`CountMatchingAsync`, + `WaitUntilReadyAsync` count-gate, `ReadAsync`, `CreateWaitReadAsync`) unchanged. +- **`ISubscriptionState` seam** driving ONE `SubscriptionRunner` loop (no fork): `SubscriptionRunnerState` + (existing stream path via `SubscribeToStream`, resume `FromStream.After(rev)`) and the new + `IndexSubscriptionState` (filtered `SubscribeToAll(FromAll, resolveLinkTos:true, + StreamFilter.Prefix($idx-user-))`, resume `FromAll.After(pos)`). `SubscriptionRunner.WithHandler` + changed only its two source-specific lines (`Subscribe()` / `Advance(e)`); dispatch, `ICaughtUpHandler`, + 5s-resubscribe-backoff, `FailFastException` are literally reused. +- **Loud guard** (SPIKE-2a/SPIKE-9 footgun): `SubscriptionRunnerState.Subscribe()` throws + `InvalidOperationException` if the stream name starts with `$idx-user-` (a direct `SubscribeToStream` there + silently delivers zero). Index streams are read ONLY via filtered `$all`. + +### S2 — opt-in wiring + PROJECTION-vs-INDEX PARITY (the acceptance bar) +- **`MergeSource` enum** (`MicroPlumberd.Services`, default `Projection`). `AddEventHandler` / + `AddSingletonEventHandler` / `AddScopedEventHandler` gained `MergeSource mergeSource = Projection`; + `EventHandlerStarter.Configure` carries it and `Start()` routes to `SubscribeEventHandlerViaIndex`. +- **`PlumberEngine.SubscribeEventHandlerViaIndex`** (public): convention output stream + `GetEventNamesFor` + → reconcile (ensure index + return the name) → subscribe via `IndexSubscriptionState` in a `SubscriptionRunner`, + same dispatch/converter as the projection path. Catch-up only. Logs handler → index → stream → types + start. +- **Registration guards (fail fast):** `UserDefinedIndex + persistently` → throw (SPIKE-7, permanent); + `UserDefinedIndex + specific-revision start` → throw (only Start/End meaningful on filtered `$all`). + +### S3 — create-new-and-swap reconciler (`UserDefinedIndexReconciler`) +- Filter-hashed name `mpidx--` (`UserDefinedIndex.IndexNameFor`, + `hash8` = first 8 lower-hex of `SHA-256(BuildFilter(types))`, ordinal-sorted ⇒ stable per set). Changed + event-type set ⇒ new name ⇒ read model rebuilds from `FromAll.Start`; unchanged ⇒ same name ⇒ 409 reuse ⇒ + no rebuild. Best-effort DELETE orphan cleanup of superseded `mpidx--*` via `ListNamesAsync` + + `DeleteAsync` (correctness never depends on cleanup running). + +### S4 — coexistence + permanent persistent guard + proliferation +- Projection-backed and index-backed handlers coexist in one engine; the projection default creates NO index; + the persistent+index registration guard is permanent; N index-backed merges all build (SPIKE-6 headroom). + +## Key decisions / deviations +- **Relocation over copy** (design-mandated): required adding `Migration → MicroPlumberd` `ProjectReference`. + This surfaced a namespace collision — `MicroPlumberd.StreamMetadata` began shadowing + `KurrentDB.Client.StreamMetadata` at an unqualified use site in `ProjectionCopier.cs`; fixed by fully + qualifying that one `new KurrentDB.Client.StreamMetadata(...)`. No behavior change (the 6 Migration + integration tests remain green = the S1-T5 relocation-parity guard). +- **CaughtUp semantics on the index path (empirical nuance, not a defect — DOCUMENTED at the opt-in surface):** + an index-backed subscription's history→live boundary tracks the `$all` position, so while the index is still + backfilling, `CaughtUp` can fire BEFORE the backfilled links arrive as "live". The guaranteed properties are + CaughtUp-fires + no-loss + commit-order — NOT the projection output-stream path's strict "all history + processed before CaughtUp" timeline. S2-T3 asserts the guaranteed properties. + **Consumer guidance (surfaced at the decision point):** because this is a per-handler opt-in, the trade-off is + documented in XML docs on `MergeSource.UserDefinedIndex`, on the `mergeSource:` parameter of + `AddEventHandler`/`AddSingletonEventHandler`/`AddScopedEventHandler`, and on + `PlumberEngine.SubscribeEventHandlerViaIndex` — a read model that uses `ICaughtUpHandler.CaughtUp()` as a + "fully caught up / now authoritative" readiness signal should NOT opt into index-backing (or must tolerate the + weaker guarantee); eventual delivery of all history + commit ordering are still guaranteed. +- **`HttpEndpoint` from settings:** core derives the management base + basic-auth from + `KurrentDBClientSettings.ConnectivitySettings.Address` (or first gossip seed) + `DefaultCredentials` — the + same node the gRPC client talks to; mirrors the existing `WaitUntilReady` pattern. +- Removed the dead `SubscriptionRunnerState.Handler` property (only ever assigned, never read). + +## Test results (all against real KurrentDB 26.1, image with `/v2/indexes`) +- **S1:** 4/4 — catch-up→live in commit order; resume from recorded `$all` position; footgun guard throws; + stream-backed path no-regression. Plus **6/6** pre-existing `UserDefinedIndexIntegrationTests` (Migration + relocation is byte-identical — S1-T5). +- **S2:** 4/4 — **PARITY (S2-T1, the acceptance bar): the same read model fed the same interleaved events via + the projection path and the index path reached IDENTICAL final state in the IDENTICAL delivery order**; + live-after-boot; ICaughtUpHandler fires; registration guards reject persistent + revision-start. +- **S3:** 3/3 — definition change → new filter-hashed index with widened content; no-op when unchanged; + superseded index DELETEd (orphan cleanup verified end-to-end, incl. `ListNamesAsync` parse + DELETE). +- **S4:** 4/4 — projection + index coexist; default creates no index; persistent+index rejected (projection + persistent untouched); 6 concurrent index-backed merges all build. +- **Regression:** whole solution builds 0 errors; existing subscription/read-model suites pass. One + timing-sensitive from-End test (`ReadModelTests.SubscribeModelFromEnd`) flaked once under parallel Docker + load but passes 3/3 in isolation — pre-existing flakiness, the refactored path is behavior-identical. + +## Review follow-ups closed (post-approval, before v1 lands) +- **Acceptance-path integration test (was compile-checked only).** Added a full-DI test + (`IndexBackedMergeReviewFollowupsTests.DiAcceptancePath`): a consumer registers + `AddSingletonEventHandler(mergeSource: MergeSource.UserDefinedIndex)`, boots a real `TestAppHost` + (`Host.StartAsync` → `EventHandlerService` → `EventHandlerStarter.Start` → + `SubscribeEventHandlerViaIndex(eh: null)`), and the DI-resolved singleton read model folds index-delivered + events to the expected state + tails a live append. Exercises the `eh == null` DI-resolution branch + (`SubscriptionRunner.WithHandler(func)`) + starter routing the other tests bypassed by passing an instance. +- **Cross-output orphan-delete collision guard (data-destructive edge).** Two distinct output streams whose names + NORMALIZE to the same managed base (e.g. `"Foo.1"` and `"Foo-1"` → `"foo-1"`) would share the + `mpidx--*` prefix, so reconciling one could DELETE the other's live index. Added + `UserDefinedIndex.RegisterManagedBase(outputStream)` — a process-wide owner registry keyed on the normalized + base — called first in `UserDefinedIndexReconciler.ReconcileAsync` (before any DELETE). A collision throws a + clear `InvalidOperationException` at reconcile; re-registering the SAME output stream is idempotent. Unit test + `ManagedBase_collision_is_rejected_same_name_is_idempotent` (no server). +- **Deferred (team-lead logged as follow-ups):** a mid-backfill CaughtUp reorder-window integration test; the + `NormalizeName` doc-charset nit. + +## Files +### Added (core `src/MicroPlumberd/`) +- `KurrentHttpEndpoint.cs` — relocated, public, `+FromSettings`. +- `UserDefinedIndex.cs` — relocated primitives + `EnsureAsync`/`DeleteAsync`/`GetFilterAsync`/`ListNamesAsync`/ + `IndexNameFor`/`ManagedNamePrefixFor`. +- `ISubscriptionState.cs` — the seam + `IndexSubscriptionState`. +- `UserDefinedIndexReconciler.cs` — `IIndexDefinitionReconciler` + `UserDefinedIndexReconciler`. +### Added (`src/MicroPlumberd.Services/`) +- `MergeSource.cs`. +### Added (tests `src/MicroPlumberd.Tests/Integration/`) +- `IndexBackedMergeS1Tests.cs`, `IndexBackedMergeS2Tests.cs`, `IndexBackedMergeS3Tests.cs`, + `IndexBackedMergeS4Tests.cs`, `IndexBackedMergeReviewFollowupsTests.cs` (DI acceptance path + collision guard). +### Changed +- `src/MicroPlumberd/SubscriptionRunner.cs` — `SubscriptionRunnerState : ISubscriptionState` (+guard, +Advance), + runner ctor `ISubscriptionState`, loop uses `Advance(e)`. +- `src/MicroPlumberd/PlumberEngine.cs` — store settings; lazy `UserDefinedIndex`/reconciler; + `SubscribeEventHandlerViaIndex` + `IsTailOnlyStart`. +- `src/MicroPlumberd.Services/EventHandlerStarter.cs` — `Configure(..., MergeSource)` + `Start` routing. +- `src/MicroPlumberd.Services/ContainerExtensions.cs` — `mergeSource` params + `ValidateMergeSource` guards. +- `src/MicroPlumberd.Migration/MicroPlumberd.Migration.csproj` — `ProjectReference` → core. +- `src/MicroPlumberd.Migration/UserDefinedIndexSource.cs` — delegates create/name/filter to core. +- `src/MicroPlumberd.Migration/ProjectionCopier.cs` — qualify `KurrentDB.Client.StreamMetadata`. +### Removed +- `src/MicroPlumberd.Migration/KurrentHttpEndpoint.cs` — relocated to core. diff --git a/docs/design-index-backed-merge-streams.md b/docs/design-index-backed-merge-streams.md new file mode 100644 index 0000000..c2685a5 --- /dev/null +++ b/docs/design-index-backed-merge-streams.md @@ -0,0 +1,534 @@ +# Design: Index-backed merge streams for MicroPlumberd read models + +Branch: `feature/mp-index-backed-merge-feasibility`. This is the implementation design that turns the CONFIRMED +feasibility (`feasibility-index-backed-merge-streams.md` §7, empirical spike `LiveIndexSubscriptionSpikes.cs`) +into shippable code. Every capability claim below traces to a spike result (cited as `SPIKE-n`) or is called out +as a **NEW empirical question** the engineer must close before the slice that depends on it merges. + +## Overview + +Give a read model the option to source its merged stream from a **KurrentDB 26.1 user-defined index** +(`$idx-user-`, read via filtered `$all`) instead of a `fromStreams([$et-…]).linkTo(outputStream)` join +projection. This is an **opt-in, additive** path: the projection-backed default is unchanged, both paths share +the same subscription loop / dispatch / error-backoff, and only handlers that explicitly ask for it move. + +**The confirmed dividing line (evidence-locked):** index-backing works for **CATCH-UP consumers** +(client-checkpointed, `$all`-`Position`) and NOT for **PERSISTENT-subscription consumers** — the KurrentDB +persistent-subscription pipeline does not resolve `$idx-user-*` links at all (SPIKE-7, clean four-way control, +partitions included). This split holds for BOTH merge shapes; it is cleaner than a shape-A/shape-B split. + +**v1 scope (this design, what the engineer builds now):** **shape-A only** — event-type/stream merges (the +`[EventHandler]` default), catch-up, **opt-in**, projection-backed stays the default. v1 slices are S1–S4. + +**v2 / fast-follow (proven-feasible, deferred — NOT built in v1):** **shape-B** per-key lookups +(`ProcessManagerClient`-style, keyed by a body field) via a **field-partitioned** index read by per-key prefix +`$idx-user-:`. Empirically confirmed (SPIKE-8 read PASS, SPIKE-11 subscribe PASS — per-key partition +pushes ONLY that key's events, catch-up and live, isolated). SPIKE-4's earlier "impossible" was an accessor bug +(`rec.data`/`rec.body` vs the correct `rec.value`). Fully specced in the "V2 / fast-follow" section so it can be +picked up later; the owner scoped it out of v1. + +## Goals / non-goals + +**Goals (v1)** +- An opt-in index-backed merge source for catch-up `[EventHandler]` read models (shape A), with the SAME + ordering, no-loss and no-double-process guarantees the projection path gives today. +- Zero change to existing apps: the default stays projection-backed; opting in is a per-handler choice. +- One subscription loop (no fork of `SubscriptionRunner`), extensible to the v2 shape-B per-key source. +- A **create-new-and-swap** filter-change lifecycle (the `mp_query_hash` equivalent) with DELETE-based orphan + cleanup — in-place index redefine is empirically impossible (SPIKE-5: HTTP 409; also FUTURE work per KurrentDB + 26.1 docs), so this is the settled, permanent lifecycle. Kept behind an `IIndexDefinitionReconciler` seam so an + in-place strategy can drop in IF/when KurrentDB ships index-definition updates — future-proofing, not built now. +- A loud guard against the `SubscribeToStream("$idx-user-…")` silent-zero-delivery footgun (SPIKE-2a). + +**Non-goals** (explicit — do not implement in v1) +- **Persistent-subscription consumers** (`SubscribeEventHandlerPersistently`, e.g. `ProcessManagerClient` + Inbox/Outbox) — stay on projections, **permanently** (SPIKE-7; the ONE permanent exclusion, both shapes). A + persistent index-subscription state is explicitly NOT designed; the registration guard rejecting + `UserDefinedIndex + persistently` is a permanent rule. +- **Shape-B per-key lookups** (field-partitioned index) — **deferred to v2 / fast-follow**, not built in v1. + Proven feasible (SPIKE-8/11); fully specced in the "V2 / fast-follow" section so it is ready to pick up. Owner + scope call, not a capability gap. +- **Snapshot handlers** (`SubscribeStateEventHandler`) index-backed — deferred; same mechanism can extend to + them later, not in this design. +- **Changing the default.** Projection-backed remains the default merge source. +- **Migrating existing projection-backed read models' data.** Opt-in creates a fresh index for opted-in + handlers only; existing projections/streams are untouched. Offline data migration of merges already exists + (`MicroPlumberd.Migration/UserDefinedIndexCopyContext`) and is out of scope here. +- **Specific-revision start positions** on an index-backed handler (only `Start`/`End` are meaningful on + filtered `$all`). + +## Architecture + +### Components + +| Component | Assembly | Responsibility | +|-----------|----------|----------------| +| `UserDefinedIndex` (relocated primitives) | `MicroPlumberd` (core) | Create/ensure an index (`EnsureAsync`), name normalisation + filter build, index-stream name, HTTP endpoint parsing. The reusable subset of today's `UserDefinedIndexSource`. | +| `ISubscriptionState` | `MicroPlumberd` (core) | Abstraction the runner loop consumes: `Subscribe()` → `StreamSubscriptionResult`, `Advance(ResolvedEvent)` → update resume position. Two implementations. | +| `StreamSubscriptionState` | `MicroPlumberd` (core) | Existing behaviour, refactored behind `ISubscriptionState`: `SubscribeToStream(outputStream, FromStream)`, resume `FromStream.After(OriginalEventNumber)`. | +| `IndexSubscriptionState` | `MicroPlumberd` (core) | NEW: `SubscribeToAll(FromAll, resolveLinkTos:true, StreamFilter.Prefix($idx-user-))`, resume `FromAll.After(OriginalPosition)`. | +| `SubscriptionRunner` (loop) | `MicroPlumberd` (core) | Unchanged control flow; now drives `ISubscriptionState` — one loop, both sources. | +| `PlumberEngine.SubscribeEventHandlerViaIndex` | `MicroPlumberd` (core) | NEW sibling of `SubscribeEventHandler`: ensure index (not projection) → subscribe via `IndexSubscriptionState` → same handler dispatch. | +| `EventHandlerStarter` (+ `MergeSource`) | `MicroPlumberd.Services` | Carries the opt-in choice; `Start()` routes to projection or index entry point. | +| `AddEventHandler(… mergeSource)` | `MicroPlumberd.Services` | Registration surface for opt-in. | +| `UserDefinedIndexSource` (offline reader) | `MicroPlumberd.Migration` | UNCHANGED public behaviour; refactored to delegate its create/ensure/name/filter to core `UserDefinedIndex` (no duplication). | + +### Assembly layering (load-bearing decision) + +Today `UserDefinedIndexSource`, `UserDefinedIndexSpec`, `KurrentHttpEndpoint` live in **`MicroPlumberd.Migration`**, +which is DOWNSTREAM of core. The live subscription path lives in core `MicroPlumberd` (`PlumberEngine`, +`SubscriptionRunner`) and `MicroPlumberd.Services` (starters). Core **cannot** reference Migration. + +Decision (**APPROVED by team-lead**): **relocate the index-lifecycle primitives into core** as +`UserDefinedIndex` — specifically +`EnsureAsync`, `NormalizeName`, `BuildFilter`, `EscapeJsString`, `IndexStream`/prefix const, `IndexNameFor` +(the hash-suffixed name), and `KurrentHttpEndpoint`. Migration's `UserDefinedIndexSource` keeps its +offline-only members (`CountMatchingAsync`, `WaitUntilReadyAsync` count-convergence gate, `ReadAsync`, +`CreateWaitReadAsync`) and calls the relocated core primitives for create/name/filter. This obeys "NEVER +DUPLICATE" (CLAUDE.md) — one implementation of index creation — and keeps the dependency arrow pointing the +right way (Migration → core). The relocation is behavior-preserving and is guarded by the existing Migration +integration tests (parity assertion in S1). + +Rationale for NOT copying: two divergent index-creation code paths (one in core, one in Migration) would drift +on filter/name rules and re-introduce the exact cross-repo hazards the workspace forbids. + +### Why the loop is shared, not forked + +`KurrentDBClient.SubscribeToStream(...)` and `KurrentDBClient.SubscribeToAll(...)` **both return +`StreamSubscriptionResult`** (verified against `KurrentDB.Client` 1.x), whose `.Messages` is an +`IAsyncEnumerable` carrying `StreamMessage.Event(ResolvedEvent)` and `StreamMessage.CaughtUp`. +So the only per-source differences are (1) which subscribe call is made and (2) how the resume position is +computed from a delivered `ResolvedEvent`. Both are hidden behind `ISubscriptionState`; the loop body in +`SubscriptionRunner.WithHandler` — dispatch via `OnEvent`, `ICaughtUpHandler.CaughtUp()`, the 5s +resubscribe-with-backoff, `FailFastException` handling — is unchanged and literally reused. This satisfies the +directive "reuse `SubscriptionRunner`/`ICaughtUpHandler`/error-backoff — don't fork the subscription loop", and +is the Open/Closed way to add a source (`docs/standards/dotnet.md`). + +```csharp +// core: the seam that lets one loop serve both sources +interface ISubscriptionState : IDisposable +{ + string StreamName { get; } // for logs; "$idx-user-" in the index case + CancellationToken CancellationToken { get; } + KurrentDBClient.StreamSubscriptionResult Subscribe(); // SubscribeToStream OR filtered SubscribeToAll + void Advance(ResolvedEvent e); // FromStream.After(rev) OR FromAll.After(pos) +} +``` + +`SubscriptionRunner.WithHandler` changes only these two lines: +```csharp +await using var sub = subscription.Subscribe(); // was: state-specific +// … on StreamMessage.Event(e): +await OnEvent(func, e, model); +subscription.Advance(e); // was: subscription.Position = FromStream.After(e.OriginalEventNumber) +``` + +### Process flow — index-backed catch-up subscription (steady state) + +```mermaid +sequenceDiagram + participant App as EventHandlerService (boot) + participant St as EventHandlerStarter + participant PE as PlumberEngine + participant UDI as UserDefinedIndex + participant KDB as KurrentDB 26.1 + participant SR as SubscriptionRunner + participant RM as Read model (IEventHandler) + + App->>St: Start() + St->>PE: SubscribeEventHandlerViaIndex(start) + PE->>UDI: EnsureAsync(name=IndexNameFor(out,filter), types) + UDI->>KDB: POST /v2/indexes/ {filter,start:true} (idempotent; 409 = reuse) + PE->>SR: run IndexSubscriptionState(FromAll.Start, StreamFilter.Prefix($idx-user-)) + SR->>KDB: SubscribeToAll(FromAll.Start, resolveLinkTos:true, filter) + KDB-->>SR: history links (commit order) [SPIKE-2b] + loop each ResolvedEvent + SR->>RM: Handle(metadata, ev) + SR->>SR: Advance → FromAll.After(e.OriginalPosition) [SPIKE-3] + end + KDB-->>SR: StreamMessage.CaughtUp → RM.CaughtUp() (if ICaughtUpHandler) + Note over KDB,SR: live tail: new matching appends pushed in commit order [SPIKE-1, SPIKE-2b] +``` + +### State — subscription lifecycle + +```mermaid +stateDiagram-v2 + [*] --> Ensuring + Ensuring --> CatchingUp: index created/confirmed + CatchingUp --> Live: StreamMessage.CaughtUp + Live --> CatchingUp: subscription dropped (resubscribe FromAll.After(lastPos), 5s backoff) + CatchingUp --> CatchingUp: dropped before caught up (resubscribe from lastPos or Start) + Live --> [*]: cancellation / dispose +``` + +Invariants: +- **Resume position is `$all Position`, in-memory only.** Held on `IndexSubscriptionState.Position` (a `FromAll`). + Advanced after each successfully-dispatched event to `FromAll.After(e.OriginalPosition!.Value)` (SPIKE-3). On + drop, resubscribe from the last advanced position → no re-processing of already-dispatched events, no gap. +- **No persisted checkpoint store exists to break.** The catch-up path in this repo never persists a checkpoint + across process restarts (`SubscriptionRunnerState.Position` is an in-memory field; boot starts from the + configured `start`). The index path mirrors this exactly. So the "checkpoint type changes from `StreamPosition` + to `Position`" (feasibility §4) is confined to the in-memory resume field — **there is no on-disk format + migration.** This de-risks binding constraint #2. +- **Boot start mapping:** `FromStream.Start → FromAll.Start`, `FromStream.End → FromAll.End`. A specific-revision + start is rejected at registration (`OPEN`/guard below), because it has no meaning on filtered `$all`. + +### Filter-change lifecycle — create-new-and-swap (the `mp_query_hash` equivalent) — RESOLVED + +SPIKE-5 settled the mechanism (evidence folded in; feasibility risk #4 closed): +- **In-place redefine is impossible** — POSTing a different filter to an existing index name returns **HTTP 409**; + `GET` confirms the stored filter never silently changes. So the projection-style `disable→update→enable` has no + index equivalent; a filter change MUST create a new index. +- **DELETE works** — `DELETE /v2/indexes/` returns **200**, so the superseded index can be retired. +- **Cost** — delete + recreate + full backfill of 10k matched events ≈ **3.7s**; backfill scales with the number + of matched events (see risk R2 — a real cost for high-volume event types). + +Design — one confirmed strategy today, behind a seam so a future in-place strategy can drop in: + +```csharp +// core: idempotent "ensure the current index exists and return the name to subscribe to". +interface IIndexDefinitionReconciler +{ + Task ReconcileAsync(string outputStream, IReadOnlySet eventTypes, CancellationToken ct); +} + +// THE settled, permanent default — in-place redefine is impossible today (SPIKE-5/409; KurrentDB 26.1 docs list +// "Updating an index definition" as FUTURE work). Filter-hashed name + create-new + DELETE-based cleanup. +sealed class CreateNewAndSwapReconciler : IIndexDefinitionReconciler { … } + +// NOT built now — a placeholder for IF/when KurrentDB ships index-definition updates. Do not implement until +// then; it CANNOT work on 26.1 (a redefine POST is 409-rejected). Documented so the seam's intent is explicit. +// sealed class InPlaceUpdateReconciler : IIndexDefinitionReconciler { … } // future — not viable on 26.1 +``` + +The seam is future-proofing only; `CreateNewAndSwapReconciler` is the sole registered implementation. Swapping to +in-place later is a one-line DI change gated on a KurrentDB feature that does not yet exist. + +- **Index name embeds the filter hash.** `IndexNameFor(outputStream, filter)` = + `mpidx--` where `hash8` = first 8 hex of `SHA-256(BuildFilter(types))` + (ordinal-sorted ⇒ stable for a set). Today's `IndexNameFor` hashes the output-stream name; this design changes + the hash INPUT to the **filter**, so a changed event-type set deterministically yields a NEW name. +- **Swap on change.** `ReconcileAsync` computes the current name; `EnsureAsync` creates it if new (KurrentDB + backfills, ~3.7s/10k) or reuses it (409 = definition unchanged, no rebuild). The subscription targets the + current-hash index and, for a newly-created one, reads from `FromAll.Start` — the read model **rebuilds** from + the new merged view (same observable effect as a projection `disable→update→enable`). +- **Orphan cleanup (DELETE-based).** After the new index is ready and the subscription has cut over, + `ReconcileAsync` deletes any `mpidx--*` index whose hash suffix ≠ the current one + (`DELETE /v2/indexes/` → 200). Multi-instance caveat: in a multi-instance deployment, delete only after + all instances are on the new definition (a rolling deploy transiently keeps both) — for a single-instance app + (the norm here) boot-time cleanup is safe. Correctness never depends on cleanup running: an un-deleted orphan + is harmless (extra storage + counts toward the ~60-index headroom, R3). +- **No-op when unchanged.** Same type set ⇒ same name ⇒ 409 reuse ⇒ nothing created, deleted, or rebuilt. + +### Guard against the SubscribeToStream footgun (binding constraint #1) + +`SubscribeToStream("$idx-user-", …)` silently delivers ZERO events (SPIKE-2a), and `ReadStream` on the +index stream is equally non-functional — SPIKE-9 confirmed this **exhaustively**: the index stream is reachable +ONLY through filtered `$all`, never as a direct stream. To make this fail LOUD instead of hanging silently, both +`StreamSubscriptionState.Subscribe()` and any direct index `ReadStream` helper assert their stream name does +**not** start with `UserDefinedIndex.IndexStreamPrefixRoot` (`"$idx-user-"`) and throw `InvalidOperationException` +pointing at this doc and at the filtered-`$all` reader. Index streams are ONLY ever read via filtered +`SubscribeToAll` / `ReadAllAsync` + `StreamFilter.Prefix`. Startup/contract assertion, not a hot-path cost. + +**Optimal recipe (standardized — use verbatim).** Live catch-up subscription: +`client.SubscribeToAll(FromAll.Start, resolveLinkTos: true, new SubscriptionFilterOptions(StreamFilter.Prefix("$idx-user-")))`; +checkpoint = `ResolvedEvent.OriginalPosition`; resume = `FromAll.After(pos)`. Bounded read: the same +`StreamFilter.Prefix` via `ReadAllAsync`. Per-key (shape B): prefix `"$idx-user-:"`. + +### Persistent-subscription consumers: permanently excluded (SPIKE-7, RESOLVED) + +Persistent-subscription consumers — e.g. `ProcessManagerClient`'s Inbox/Outbox merge +(`SubscribeEventHandlerPersistently`) — **cannot be index-backed, permanently.** This holds for BOTH shapes: it +is a property of the persistent pipeline, not of the merge shape. KurrentDB limitation, not a usage bug (SPIKE-7, +real KurrentDB 26.1, with controls): + +- A persistent group over `StreamFilter.Prefix("$idx-user-")` delivers **ZERO** on catch-up AND live — + `resolveLinkTos` both **true and false**; a persistent `CreateToStream` directly on `$idx-user-` also **0**. +- **Control A** (persistent filtered-`$all` over a normal `EventTypeFilter`) delivered **3/3** → harness sound, + the zero is real. The catch-up `SubscribeToAll`+filter path DOES resolve the links (**5/5**) — the persistent + pipeline simply does not resolve `$idx-user-*` links. + +Consequences: the `mergeSource: UserDefinedIndex` + `persistently: true` registration guard is a **permanent +rule**; **no persistent index-subscription state is designed or built**; the `ISubscriptionState` seam has one +family only — catch-up (`StreamSubscriptionState`, `IndexSubscriptionState`). + +## APIs / Protocols + +### Registration (opt-in) + +```csharp +public enum MergeSource { Projection = 0, UserDefinedIndex = 1 } // default = Projection (safe) + +// New overload / optional parameter on the existing registration surface: +services.AddEventHandler(mergeSource: MergeSource.UserDefinedIndex); + +// Unchanged default — still projection-backed, nothing about existing apps changes: +services.AddEventHandler(); +``` + +Registration-time guards (fail fast, `docs/standards/dotnet.md`): +- `mergeSource: UserDefinedIndex` + `persistently: true` ⇒ `throw InvalidOperationException` + ("index-backed merge supports catch-up subscriptions only; persistent subscriptions stay projection-backed"). +- `mergeSource: UserDefinedIndex` + a specific-revision `start` ⇒ `throw` ("only Start/End are valid on an + index-backed handler"). + +### `PlumberEngine` entry point + +```csharp +public async Task SubscribeEventHandlerViaIndex( + TEventHandler? eh = null, + string? outputStream = null, // convention-derived if null (same as projection path) + FromRelativeStreamPosition? start = null, // Start (default) or End only + CancellationToken token = default) + where TEventHandler : class, IEventHandler, ITypeRegister; +``` + +Behaviour: derive `outputStream` via `Conventions.OutputStreamModelConvention`; derive the event-type set via +`_typeHandlerRegisters.GetEventNamesFor()`; `name = await reconciler.ReconcileAsync(outputStream, +types, token)` (`UserDefinedIndexReconciler` ensures the current filter-hashed index exists — create-new-and-swap ++ DELETE cleanup on change, no-op when unchanged — and returns the name to target); then subscribe through an +`IndexSubscriptionState` over `$idx-user-` wired into a `SubscriptionRunner` with the same +`WithHandler`/converter used by `SubscribeEventHandler`. Catch-up only. + +### Index HTTP contract (reused verbatim from `UserDefinedIndexSource`) + +Create (idempotent; 409 = reuse): +``` +POST /v2/indexes/ +{ "filter": "rec => rec.schema.name == \"FooCreated\" || rec.schema.name == \"FooUpdated\"", "start": true } +``` +Read stream: `$idx-user-`, consumed ONLY via `SubscribeToAll(FromAll, resolveLinkTos:true, +StreamFilter.Prefix("$idx-user-"))`. + +## Dependencies + +| Dependency | Purpose | Failure handling | +|------------|---------|------------------| +| KurrentDB 26.1 user-defined indexes (`/v2/indexes`, `$idx-user-*`) | The merge source | Create is idempotent (409 reuse). If the node predates 26.1 the POST fails → `EnsureAsync` throws with the failing URI logged; the handler fails to start (loud), it does not silently fall back to a projection. | +| `KurrentDBClient.SubscribeToAll` + `StreamFilter.Prefix` | Live catch-up→tail delivery | Same `catch (Exception) → 5s resubscribe` as the stream path; resubscribe from the in-memory `FromAll` position (no gap/dup). `FailFastException` still bubbles to shut the app down. | +| `UserDefinedIndex` (relocated to core) | One index-lifecycle implementation shared by live + offline paths | Behavior-preserving relocation; Migration parity test (S1) guards it. | +| `MicroPlumberd.Testing` `EventStoreServer` (Docker, KurrentDB 26.1) | Integration test substrate | Per workspace rule, NEVER Testcontainers; each test gets an isolated in-memory server. | + +## Vertical-slice implementation plan (v1 = shape-A, slices S1–S4) + +**v1 scope is S1–S4 only.** Shape-B is a v2 fast-follow (spec in the "V2 / fast-follow" section, not a v1 slice). +Each slice is independently buildable (`dotnet.exe build src/MicroPlumberd.sln`) and ends with a green +integration test against a **real KurrentDB 26.1** via `MicroPlumberd.Testing` (`EventStoreServer.StartInDocker`). +Commit at each slice boundary (`docs/standards/dev-process.md`; team-lead owns git). + +### S1 — Index-backed subscription primitive (core) +Relocate `UserDefinedIndex` primitives into core + refactor Migration to delegate; add `ISubscriptionState`, +`StreamSubscriptionState` (behaviour-preserving), `IndexSubscriptionState` (filtered `SubscribeToAll`, +`FromAll` resume); refactor `SubscriptionRunner.WithHandler` to drive `ISubscriptionState`; add the +`$idx-user-` `SubscribeToStream` guard. + +Integration tests: +- **S1-T1 (catch-up→live parity with the spike, through the framework abstraction):** create an index over + types A,B; run `IndexSubscriptionState` inside a `SubscriptionRunner`; assert history `0,1,2` then live-append + `3,4` arrive in commit order — the SPIKE-2b guarantee, now exercised through production types. +- **S1-T2 (resume):** record position after `0,1,2`, dispose, append `3,4`, resubscribe from the recorded + `FromAll` position; assert only `3,4` delivered (SPIKE-3). +- **S1-T3 (footgun guard):** `StreamSubscriptionState.Subscribe()` on a `$idx-user-…` name throws + `InvalidOperationException` (never a silent zero-delivery). +- **S1-T4 (no regression):** existing stream-backed `SubscriptionRunner` path still delivers in order + (control — the refactor changed the seam, not the behaviour). +- **S1-T5 (Migration parity):** existing `UserDefinedIndexSource` offline read still passes after delegating to + core `UserDefinedIndex` (guards the relocation). + +### S2 — Opt-in wiring on one read model + projection parity +Add `MergeSource` to `EventHandlerStarter.Configure` + `AddEventHandler(mergeSource:)`; add +`PlumberEngine.SubscribeEventHandlerViaIndex`; registration guards. + +Integration tests: +- **S2-T1 (the load-bearing correctness test — parity):** register the SAME `[EventHandler]` read model twice + in one app against real KurrentDB — once `MergeSource.Projection`, once `MergeSource.UserDefinedIndex` — feed + an identical interleaved event stream, assert both read models reach **identical state** and observed the + **same delivery order**. This is the concrete form of "exactly the ordering + no-loss + no-double-process the + projection path gives today." +- **S2-T2 (live after boot):** with the index-backed handler running, append new matching events post-boot; + assert they are delivered live in commit order (SPIKE-1/2b end-to-end). +- **S2-T3 (`ICaughtUpHandler`):** an index-backed handler implementing `ICaughtUpHandler` receives `CaughtUp()` + after history — regression-guards `OPEN-4` (PASS). +- **S2-T4 (guards):** `UserDefinedIndex + persistently` and `UserDefinedIndex + revision-start` throw at + registration. + +### S3 — Create-new-and-swap on definition change (lifecycle RESOLVED by SPIKE-5) +`UserDefinedIndexReconciler.ReconcileAsync`: filter-hashed name (`mpidx--`), create-new + +`FromAll.Start` rebuild on change, DELETE-based orphan cleanup after cut-over. No strategy seam (in-place ruled +out, SPIKE-5/409). + +Integration tests: +- **S3-T1 (definition change → correct merged view):** boot handler v1 (types {A}); feed events; boot v2 (types + {A,B}); assert a new `mpidx-…-` index exists, the read model now includes B events, and there is no + double-processing within a run. +- **S3-T2 (no-op when unchanged):** re-boot with the same type set ⇒ same name, 409 reuse, no rebuild, no delete. +- **S3-T3 (orphan cleanup):** after a swap, assert the old-hash `mpidx-…-` index is DELETEd (200) and + only the current-hash index remains; a read from the deleted index no longer resolves. + +### S4 — Coexistence + persistent guard + proliferation +Confirm the default, the permanent persistent exclusion, and the ceiling headroom. + +Integration tests: +- **S4-T1 (coexistence):** one projection-backed and one index-backed handler in the same app boot both build + correct read models — the two mechanisms run side by side. +- **S4-T2 (default unchanged):** a handler registered with no `mergeSource` creates a join projection and no + index (assert `$idx-user-*` absent for it). +- **S4-T3 (persistent stays on projections):** `ProcessManagerClient`'s Inbox/Outbox + (`SubscribeEventHandlerPersistently`) is unchanged and never index-backed; the `UserDefinedIndex + persistently` + registration guard throws (permanent rule, SPIKE-7). +- **S4-T4 (ceiling headroom — OPEN-2 resolved):** SPIKE-6 already confirmed 60 indexes/node with no degradation + (3× today's count). Regression guard, not a gate — assert N index-backed handlers in one app all build + correctly; re-measure a hard ceiling only if the live count approaches ~60. + +The v1 slice list is S1–S4 (above). Shape-B is a v2 fast-follow — spec below, NOT a v1 slice. + +## V2 / fast-follow — shape-B field-partitioned per-key lookup (PROVEN-FEASIBLE, deferred) + +> **NOT part of v1.** The owner scoped v1 to shape-A. This section is a proven-feasible spec (SPIKE-8 read PASS, +> SPIKE-11 subscribe PASS — per-key partition pushes ONLY that key's events, catch-up and live, isolated) so v2 +> can pick it up without re-discovery. Index-backable for **catch-up** consumers only; the persistent exclusion +> applies here too. + +`EnsureLookupProjection` today fans a `$ce-` stream out into one stream per key via +`linkTo('-' + e.body., e)`. SPIKE-8/11 show a KurrentDB 26.1 **field-partitioned index** reproduces +this for catch-up consumers. + +**Scalability (corrects the earlier proliferation concern):** shape-B needs **ONE field-keyed index per lookup +category — NOT one index per key value.** KurrentDB fans that single index out into per-value partition streams +(`$idx-user-:`) itself. A lookup over N distinct keys is ONE index, not N — no unbounded index +proliferation. (Field config allows max 1 field/index — exactly one routing property, matching +`EnsureLookupProjection`'s single-key design.) + +Index definition (adds a `fields` selector; body accessor is `rec.value`, NOT `rec.data`/`rec.body` — the SPIKE-4 bug): +``` +POST /v2/indexes/ +{ "filter": "rec => rec.schema.name == 'X'", + "fields": [ { "name": "recipientid", "selector": "rec => rec.value.RecipientId", "type": "INDEX_FIELD_TYPE_STRING" } ], + "start": true } +``` +Per-key read/subscribe — the key is a **prefix suffix** on the index read stream: +```csharp +var filter = StreamFilter.Prefix($"$idx-user-{name}:{key}"); // bounded per-key read (Rehydrate analog) +client.ReadAllAsync(Direction.Forwards, Position.Start, filter, resolveLinkTos: true); +client.SubscribeToAll(FromAll.Start, resolveLinkTos: true, // live per-key catch-up + new SubscriptionFilterOptions(StreamFilter.Prefix($"$idx-user-{name}:{key}"))); +``` + +v2 work items: `UserDefinedIndex` create gains an optional `fields` selector; `UserDefinedIndexSource` gains +`ReadPartitionAsync(name, key)` (bounded) plus an optional live per-key `IndexSubscriptionState` with a `:` +prefix. Everything else — filter-hashed name, create-new-and-swap, the `SubscribeToStream`/`ReadStream` guard — +carries over unchanged. + +**`$ce`-category source — RESOLVED (SPIKE-12, no caveat).** `EnsureLookupProjection` sources from +`$ce-` (every event in a stream category). An index expresses this DIRECTLY via +`rec.position.stream.startsWith("-")` — proven with a control (same event type in two categories: a +type-only filter matches both, the category filter matches only the target). So it is a straight translation of +`fromStreams(['$ce-{category}'])`, NOT an enumerated-OR-of-event-types workaround (no "misses a later-added type" +risk). Consequence: the v2 `EnsureFieldPartitionedIndexAsync` base predicate can be EITHER an event-type filter +(`rec.schema.name == …`, shape A) OR a category filter (`rec.position.stream.startsWith("-")`, shape B's +`$ce` source), combined with the optional single `fields` selector for per-key partitioning (max 1 field — +matches `EnsureLookupProjection`'s single `eventProperty`). Shape-B guidance is now fully closed, no open items. + +#### Definitive `ProcessManagerClient` determination (from the code — answers the v2 routing question) + +`ProcessManagerClient.SubscribeProcessManager` (`MicroPlumberd.Services.ProcessManager/ProcessManagerClient.cs`) +has **two consumption paths** that resolve OPPOSITELY: + +| Concern | Code | Consumption | Index-backable? | +|---------|------|-------------|-----------------| +| Inbox / Outbox **merge** | lines 103–104: `SubscribeEventHandlerPersistently(sender,"…Outbox")` / `(executor,"…Inbox")` | **PERSISTENT** | **NO — permanent** (SPIKE-7). Stays on projections. | +| `{PM}Lookup` **per-key lookup** | line 107 `EnsureLookupProjection(…,"RecipientId","…Lookup")`; consumed in `GetManager` line 129 `_plumber.Rehydrate(lookup,"…Lookup-{recipientId}")` | **bounded catch-up READ** (`Rehydrate` = `ReadStream`) | **YES — via field partitions** (v2). | + +So: the PM **merge** stays on projections (persistent); the PM **lookup** is a catch-up `Rehydrate` read → a v2 +index-backing candidate. The one required rewire is the read — `Rehydrate` uses `ReadStream("{PM}Lookup-{id}")`, +and `ReadStream`/`SubscribeToStream` on an index stream is definitively non-functional (SPIKE-9), so the lookup +read becomes a filtered-`$all` prefix read of `$idx-user-:{recipientId}`. That `GetManager`/`Rehydrate` +rewire is a v2 item and optional — the PM lookup keeps working on its projection until migrated. + +v2 slice sketch (integration tests, when v2 is scheduled): partition read isolation (`:` yields only x, commit +order — SPIKE-8); partition live tail (SPIKE-11); `rec.value` accessor works / `rec.data`·`rec.body` index +nothing (guards the SPIKE-4 regression); `ReadStream`/`SubscribeToStream` on `$idx-user-:` throws +(SPIKE-9). + +## Correctness guarantees (mapped from projection path to index path) + +| Property | Projection path (today) | Index path (this design) | Evidence | +|----------|-------------------------|--------------------------|----------| +| Commit ordering | Output-stream link order == commit order | Filtered-`$all` link order == commit order, incl. live appends | SPIKE-1, SPIKE-2b | +| Catch-up → live handoff | `SubscribeToStream` history then tail; `StreamMessage.CaughtUp` | `SubscribeToAll`+filter history then tail; `StreamMessage.CaughtUp` fires at the exact history→live boundary, driving `ICaughtUpHandler` | SPIKE-2b + OPEN-4 (PASS) | +| No loss / no double-process on drop | resubscribe `FromStream.After(rev)` | resubscribe `FromAll.After(pos)` | SPIKE-3 | +| Restart resume | in-memory position; boot from configured start | identical, in-memory `FromAll`; boot from Start/End | design + SPIKE-3 | +| Definition change | `mp_query_hash` disable→update→enable, read model replays | filter-hashed name ⇒ new index (in-place impossible), read model rebuilds, old index DELETEd | SPIKE-5 (409 in-place / 200 DELETE / ~3.7s per 10k) | + +## Risk register + +| # | Risk | Bite | Mitigation / status | +|---|------|------|---------------------| +| R1 | Silent zero-delivery via `SubscribeToStream("$idx-user-…")` | A future implementer "simplifies" the index path to a stream subscribe → silent hang | Loud guard in `StreamSubscriptionState.Subscribe()` + this doc + S1-T3. SPIKE-2a. | +| R2 | Every event-type-set change triggers a FULL re-backfill (in-place redefine impossible, SPIKE-5/409) | Real latency/cost for high-volume event types — ~3.7s/10k matched events, grows with matched count; the read model is rebuilding meanwhile | Inherent to the confirmed lifecycle. Mitigate by making event-type-set changes rare (they already are — a code change), and by keeping the read model available on the OLD index until the new one is caught up (cut over only after catch-up). DELETE-based cleanup keeps orphan count bounded. | +| R3 | Index count ceiling — headroom confirmed at 60, hard ceiling still unmeasured | Naive 1:1 index-per-read-model (20 today, growing) + transient orphans during swaps | SPIKE-6: 60 indexes created in ~1.3s, 0 rejections, no early/late degradation under concurrent backfill — comfortable at 3× today's count. Hard ceiling not probed; opt-in + DELETE cleanup keep the live count near the read-model count. Re-measure only if count approaches ~60. | +| R4 | `StreamMessage.CaughtUp` on filtered `$all` — RESOLVED | — | OPEN-4 PASS: fires at the exact history→live boundary; `ICaughtUpHandler` works. Regression-guarded in S2-T3. | +| R5 | Two parallel merge mechanisms increase ops surface | Debugging/observability | Accepted as the price of a non-breaking opt-in; log which source each handler uses at start (proper-logging standard). Revisit if index-backed ever becomes default. | +| R6 | Relocating `UserDefinedIndex` to core regresses the offline migration path | Migration behaviour drift | Layering APPROVED by team-lead (Migration→core arrow correct). Behavior-preserving relocation guarded by S1-T5 parity test. | +| R7 | Offline count-convergence readiness gate misapplied to a LIVE, growing store | `WaitUntilReadyAsync(expectedCount)` would hang forever live | RESOLVED — OPEN-6 PASS: the live path subscribes to filtered `$all` immediately after `EnsureAsync` and tails, NO count gate. The offline gate is used only by the Migration bounded read. | + +## Empirical questions — ALL RESOLVED (evidence folded into the design) + +Nothing is left open; the design is fully evidenced. + +- **OPEN-1 / risk #4 — filter-change lifecycle + DELETE — RESOLVED.** In-place redefine impossible (SPIKE-5: 409, + `GET` confirms no silent change; KurrentDB 26.1 docs list index-definition updates as FUTURE work); + `DELETE /v2/indexes/` = 200; delete+recreate+backfill 10k ≈ 3.7s. ⇒ create-new-and-swap + DELETE cleanup. +- **OPEN-3 — in-place mutation impossible — RESOLVED (yes).** Same SPIKE-5/409. `IIndexDefinitionReconciler` seam + retained only as future-proofing. +- **OPEN-2 — index count ceiling — RESOLVED (headroom).** SPIKE-6: 60 indexes/node in ~1.3s, 0 rejections, no + degradation. Comfortable at 3× today's 20; hard ceiling unmeasured (R3). +- **OPEN-4 — `StreamMessage.CaughtUp` on filtered `SubscribeToAll` — RESOLVED (PASS).** Fires at the exact + history→live boundary; `ICaughtUpHandler` fires on it. Regression-guarded in S2-T3. +- **OPEN-6 — LIVE path uses NO count-based readiness gate — RESOLVED (PASS).** Subscribe-then-tail is loss-free; + the offline `WaitUntilReadyAsync(expectedCount)` gate is used only by the Migration bounded read. +- **OPEN-5 / SPIKE-7 — persistent-subscription over filtered `$all` — RESOLVED (permanent NO).** Persistent group + over `$idx-user-*` (resolveLinkTos true/false) and persistent `CreateToStream` both deliver ZERO; Control A + (persistent filtered-`$all` over `EventTypeFilter`) = 3/3 → real KurrentDB limitation. Persistent consumers stay + projection-backed permanently (the one permanent exclusion). +- **OPEN-7 / SPIKE-8/11 — shape-B via field partitions — RESOLVED (PASS, rescued).** SPIKE-4's "impossible" was an + accessor bug (`rec.data`/`rec.body` vs the real `rec.value`). A field-partitioned index + `:`-prefixed + filtered-`$all` read (SPIKE-8) with confirmed per-key subscribe isolation (SPIKE-11) reproduces per-key lookup + for catch-up consumers. ⇒ shape B is index-backable (v2 fast-follow). +- **OPEN-8 / SPIKE-12 — `$ce`-category source as an index filter — RESOLVED (PASS).** + `rec.position.stream.startsWith("-")` selects by category directly (proven with a control). ⇒ the v2 + base predicate can be event-type OR category; shape-B guidance is fully closed, no caveat. +- **OPEN-9 / SPIKE-9 — direct index-stream access — RESOLVED (definitively non-functional).** Both + `SubscribeToStream` and `ReadStream` on `$idx-user-*` deliver nothing, exhaustively. ⇒ the loud guard stays and + covers reads too; index streams are reached ONLY via filtered `$all`. + +## Implementation notes + +- **Simplify before optimizing / delete before adding** (`docs/PRINCIPLES.md`): the loop is reused via one small + interface rather than a second runner; index creation is relocated, not copied. +- **Proper logging at every seam** (workspace standard): `EnsureAsync` already logs create/reuse + failing URI; + `SubscribeEventHandlerViaIndex` logs `handler → index name → index stream → event types`, and the starter logs + the chosen `MergeSource` at start so "which path is this read model on?" self-reports in the log. +- **Fail fast** (`docs/standards/dotnet.md`): registration guards throw; a missing/old KurrentDB makes the + handler fail to start loudly (never a silent projection fallback). +- **No magic strings:** `"$idx-user-"` is `UserDefinedIndex.IndexStreamPrefixRoot`; the index name pattern is + built by `IndexNameFor`. + +## Validation Status +- [x] Completeness (event-modeling): every v1 opt-in intention → index-ensure + filtered-`$all` subscription → + read-model view is traced; v1 scope = shape-A catch-up (S1–S4); shape-B is a proven-feasible v2 fast-follow; + persistent consumers are the one permanent exclusion (evidence); snapshot handlers deferred. +- [x] Clarity: no vague terms; each guarantee is quantified or cited to a spike. +- [x] Consistency: one index-creation implementation; terminology matches the feasibility doc and code. +- [x] Testability: every v1 slice maps to a named integration test against real KurrentDB 26.1; the parity test + (S2-T1) is the acceptance bar. +- [x] Dependencies: KurrentDB index API, `SubscribeToAll`, relocated `UserDefinedIndex`, `MicroPlumberd.Testing` + documented with failure handling. +- [x] Conflicts: coexistence (not replacement) resolves the default-safety vs new-capability tension; layering + decision resolves the core↔Migration dependency direction. + +**v1 is FINAL — no open v1 items.** All empirical questions (OPEN-1..9, 12 spikes across 3 rounds) are RESOLVED with spike evidence folded +in. The confirmed dividing line: catch-up consumers index-backable (both shapes), persistent consumers +projection-only (both shapes). v1 ships shape-A (S1–S4); shape-B is a proven-feasible v2 fast-follow, now fully +closed with no open design items (`$ce`-category source resolved by SPIKE-12). diff --git a/feasibility-index-backed-merge-streams.md b/feasibility-index-backed-merge-streams.md new file mode 100644 index 0000000..022b191 --- /dev/null +++ b/feasibility-index-backed-merge-streams.md @@ -0,0 +1,421 @@ +# Feasibility framing — KurrentDB 26.1 user-defined index as a LIVE replacement for merge/output streams + +Branch: `feature/mp-index-backed-merge-feasibility`. Analysis only — no production code changed by this doc. + +Question from the owner: can the `fromStreams([...]).linkTo(outputStream)` JS join projection that sits +behind read models be replaced by a KurrentDB 26.1 **user-defined index**? This framing maps the current +machinery precisely and states what a replacement must satisfy. + +**VERDICT (updated twice after `eng-mergefeas`'s empirical spikes — see §7/§8): YES for shape (A) AND shape (B), +both catch-up-subscription-only.** A live, commit-ordered, subscribable index-backed merge stream is achievable +on KurrentDB 26.1 for the multi-`$et`-join / single-output-stream pattern — but only via `SubscribeToAll` + +`StreamFilter.Prefix` with `$all`-`Position` checkpointing (NOT a direct `SubscribeToStream` against the index's +read stream, which silently returns nothing — reconfirmed for shape A in §8 SPIKE-9). Body-property lookup/fan-out +projections (shape B) were first found FAIL (§7 SPIKE-4) but that verdict is **SUPERSEDED**: SPIKE-4 used the +wrong body accessor (`rec.data`/`rec.body`); the correct accessor is `rec.value`, and a dedicated `"fields"` +index config partitions matching events into per-value streams (`$idx-user-:`) — the index-native +equivalent of `EnsureLookupProjection`'s routing (§8 SPIKE-8). The hard, *permanent* boundary for **both** shapes +is the subscription mechanism, not the filter: **persistent subscriptions cannot see index links at all** (§7 +SPIKE-7), so any handler using `SubscribeEventHandlerPersistently` — shape A or B — stays projection-backed +regardless of what the filter can express. §7/§8 have the raw evidence; §1/§2/§4/§5/§6 below are revised to +reflect it. (Lesson for future readers of this doc: treat any single spike's FAIL as provisional until a second +pass with the right API surface has been tried — SPIKE-4's FAIL held for one full round of reporting before +SPIKE-8 overturned it.) + +## 1. Inventory — how many merge-stream shapes exist, and how load-bearing they are + +Two distinct query shapes are generated by `KurrentDBProjectionManagementClientExtensions` +(`src/MicroPlumberd/EventStoreProjectionManagementClientExtensions.cs`): + +**(A) Multi-`$et` join → single output stream** — `CreateQuery` (line 168): +```js +fromStreams(['$et-TypeA','$et-TypeB']).when({ $any: function(s,e){ linkTo('OutputStream', e) } }); +``` +Built by `TryCreateJoinProjection` (two overloads). This is the **default path for every `[EventHandler]` +read model** registered the standard way — `SubscribeEventHandler`, `SubscribeEventHandlerPersistently`, +`SubscribeStateEventHandler` (`PlumberEngine.cs:209-369`) all call `TryCreateJoinProjection` with +`ensureOutputStreamProjection: true` by default. Even a handler with **one** event type goes through this +(a 1-element `fromStreams([...])`) — the join projection is not conditional on ≥2 types. `AddEventHandler()` +(`MicroPlumberd.Services/ContainerExtensions.cs`) wires this into `EventHandlerService`, a `BackgroundService` +that calls `Start()` → `SubscribeEventHandler`/`SubscribeEventHandlerPersistently` on every +registered handler at app boot. Repo-wide, 20 classes carry `[EventHandler]`/`[OutputStream(...)]`; there is no +opt-out short of explicitly passing `ensureOutputStreamProjection: false`, which no call site in this repo does. +**"Behind almost every read model" is accurate for shape (A)** — it is the framework's default, not an +opt-in feature. + +**(B) `$ce` category + body-property fan-out → N dynamically-named streams** — `EnsureLookupProjection` +(line 40): +```js +fromStreams(['$ce-{category}']).when({ $any: function(s,e){ + if(e.body && e.body.{eventProperty}) linkTo('{outputStreamCategory}-' + e.body.{eventProperty}, e) +}}); +``` +Only one caller in the repo: `ProcessManagerClient.SubscribeProcessManager` (`MicroPlumberd.Services.ProcessManager/ProcessManagerClient.cs:106-108`), building `{ProcessManager}Lookup` over the process-manager's `Inbox`/`Outbox` +category, keyed by `RecipientId`. This is **not one merge stream** — it's a projection that fans a category out +into an unbounded number of per-key streams (`FooLookup-`), each read once via `Rehydrate` +(one-shot catch-up read to build a lookup, not a live subscription — see `ProcessManagerClient.GetManager`, +line 124-138). Low cardinality in this repo (1 use site) but structurally the harder case — **and, per the §8 +reversal below, this one call site is the best-fit candidate in the repo for a shape-B index swap**: its +consumption pattern is `Rehydrate` (a bounded, one-shot forward read — catch-up only, never live-tailed), which +is exactly the subscription shape the field-partition index supports (§8 SPIKE-8). Note this is a *distinct* use +from the SAME process manager's `Inbox`/`Outbox` (shape A, consumed via `SubscribeEventHandlerPersistently` — +excluded regardless of shape by §7 SPIKE-7): the Lookup and Inbox/Outbox streams for one process manager could, +in principle, end up on different mechanisms (Lookup → index, Inbox/Outbox → projection) even though they're +wired by the same `SubscribeProcessManager` call today. + +Existing discovery code (`MicroPlumberd.Migration/ProjectionCopier.cs`) only understands shape (A): `ParseLinkTypes` +matches `'\$et-([^']+)'` and `ParseOutputStream` matches a literal `linkTo('...'` — both regexes **fail to match** +shape (B) (its `linkTo` target is a concatenated expression, its source is `$ce-`, not `$et-`). Confirmed in +`UserDefinedIndexMergeBuilder.BuildAsync` (`UserDefinedIndexCopyContext.cs:85-89`): when `LinkTypes.Count == 0` it +logs and **skips index creation** for that projection. So today's index-migration path already silently excludes +shape (B) — it was never built to cover it. + +## 2. Capabilities a user-defined index must provide to replace a merge stream behind a LIVE read model + +The existing `UserDefinedIndexSource` (`MicroPlumberd.Migration/UserDefinedIndexSource.cs`) is a **bounded, +one-shot historical reader** built for an offline migration: create → poll until the indexed count equals a +known, frozen total → read once. None of its four capabilities below are exercised by it today; each is a +distinct thing the live spike must prove separately. + +**(a) Filter expressivity — CORRECTED post-spike (see §8).** `BuildFilter` (line 276-283) emits +`rec => rec.schema.name == "A" || rec.schema.name == "B"`, a pure event-type-name predicate — this expresses +shape (A) exactly (event-type set ↔ OR-of-equality) and was never in question. What *was* wrong, in the first +spike round (§7 SPIKE-4) and in this framing's original text here, was the assumption that shape (B) is +unreachable: that spike probed `rec.data.*`/`rec.body.*` (both wrong accessors) and found nothing, which looked +like confirmation the filter can't see event bodies at all. §8 SPIKE-8 shows the correct accessor is `rec.value`, +and — more importantly — that filter predicates aren't the whole story: KurrentDB 26.1 has a separate `"fields"` +index config (`{"name":"...", "selector":"rec => rec.value.SomeId", "type":"INDEX_FIELD_TYPE_STRING"}`, capped at +**one field per index** per the KurrentDB docs) that materializes a *partition stream per distinct field value* +(`$idx-user-:`) — this is the index-native analog of `EnsureLookupProjection`'s +`linkTo('cat-' + e.body.RecipientId, e)` fan-out. So the filter is still a WHERE (unchanged), but the index as a +whole is NOT purely a WHERE-then-single-stream primitive — the fields config gives it a GROUP-BY-into-streams +capability this framing didn't know existed when first written. The one-field cap means only a *single* routing +property is expressible (matches `EnsureLookupProjection`'s single `eventProperty` parameter exactly — no +regression for the one real use site in this repo, but a hard ceiling on any future multi-property routing key). + +**(b) Live catch-up + tail subscription to `$idx-user-`.** Every read-model subscription path in this repo +(`SubscriptionRunner`/`SubscriptionSeeker` in `MicroPlumberd/SubscriptionRunner.cs`, plus the persistent-subscription +path in `SubscriptionSet.cs`) is built on `KurrentDBClient.SubscribeToStream` / `KurrentDBPersistentSubscriptionsClient.SubscribeToStream` +against a regular stream, catch-up (`FromStream.Start`/`After(pos)`) then indefinite live tail inside a +`Task.Factory.StartNew(..., LongRunning)` loop, with `StreamMessage.CaughtUp` driving `ICaughtUpHandler.CaughtUp()` +(`SubscriptionRunner.cs:179-183`) and automatic resubscribe-with-backoff on drop (`catch (Exception)` → 5s retry, +`SubscriptionRunner.cs:204-208`). `UserDefinedIndexSource.ReadAsync` uses `ReadAllAsync` + `StreamFilter.Prefix` +— a **finite forward read of `$all`**, not `SubscribeToAll`/`SubscribeToStream`. Whether a `$idx-user-*` stream (or +the filtered-`$all` read underneath it) supports the client's subscribe-catch-up-then-tail primitive at all is +unproven in this codebase — `UserDefinedIndexSource` has never called a subscribe API, only `ReadAllAsync`. + +**(c) Checkpoint/resume.** The persistent-subscription path (`SubscriptionSet.SubscribePersistentlyAsync`) relies +on server-side checkpointing (`PersistentSubscriptionSettings(checkPointLowerBound: minCheckpointCount)`) so a +crashed consumer resumes from its group's last ack. The catch-up path relies on `FromStream.After(lastEventNumber)` +computed client-side (`SubscriptionRunner.cs:177`, position tracked per delivered event). Either mechanism assumes +stable, resumable event numbers on the subscribed stream. A `$idx-user-*` read stream would need an equivalent +resumable position (its own revision numbers, or an `$all`-Position pair) that both the plain catch-up path and a +persistent-subscription-style group checkpoint could resume against. `UserDefinedIndexSource` never resumes a +partial read — it always starts at `Position.Start` (line 235) — so resumability is unproven, not just untested. + +**(d) Lifecycle / proliferation / change-detection.** Shape (A) uses `mp_query_hash` (SHA-256 of the generated +query, stamped as custom stream metadata on the output stream — `EventStoreProjectionManagementClientExtensions.cs:129-166`) +so `TryCreateJoinProjection` is a no-op when the handler's event-type set hasn't changed, and safely +disable→update→enable when it has. An index has no update-in-place: `EnsureAsync`'s own doc says "KurrentDB does +not redefine an index in place" — a changed filter on an existing index name is left as-is with a logged warning +(`UserDefinedIndexSource.cs:43-44`, `WarnOnFilterDriftAsync`). A drop-in replacement needs an equivalent +change-detection guard, but the underlying primitive (mutate a live index's filter) may not exist at all — this +could mean "delete + recreate + rebuild from Start" on every handler code change, which is a materially different +operational cost than the current disable/update/enable of a projection. One index per `[OutputStream]` read +model (20 in this repo) is the naive 1:1 mapping for shape (A) — **corrected from the original text here**: shape +(B) is NOT one index per distinct `RecipientId` value (that would indeed be unbounded and clearly unworkable); +per §8 SPIKE-8, it is **one index total per lookup category**, configured with a single `"fields"` partition +key, which KurrentDB itself fans out into per-value streams (`$idx-user-:`) as data arrives — the +index count stays at "one per `[OutputStream]`-or-lookup-category", not "one per key". Whether KurrentDB has a +practical ceiling on live index count remains open (§7 SPIKE-6 only tested 60 whole-index, non-partitioned, +instances). + +## 3. Empirical questions for the spike (pass/fail criteria) + +Frame each as binary; the spike should report PASS/FAIL/UNKNOWN-BLOCKED per item against a real KurrentDB 26.1 +node, not against documentation. + +1. **Live-updated + commit-ordered under tail.** After `WaitUntilReadyAsync` confirms backfill-complete, append + N more matching events to `$all` (mixed types, matching the filter) *while a reader is attached*. FAIL if the + index stops advancing, advances out of commit order, or requires a poll/re-read cycle instead of pushing new + entries. (The bounded-read spike already proved commit order for the *backfill*; this must reprove it for + *steady-state appends*, which is a different code path in KurrentDB.) +2. **Subscribability.** Does `KurrentDBClient` expose (or can it be coerced into) a catch-up subscription against + `$idx-user-` — either `SubscribeToStream("$idx-user-", ...)` directly, or `SubscribeToAll` + + `StreamFilter.Prefix` — that delivers history then transitions to live push, the same shape `SubscriptionRunner` + requires? FAIL if the only available primitive is a poll-a-finite-read-repeatedly loop (that would be a + regression from push-based catch-up to client-side polling for every read model in the framework). +3. **Resumable position survives restart.** Kill and restart the subscriber mid-stream; does resubscribing from + the last-seen position (revision, or `$all` Position pair) replay exactly the missed events once, with no + gap and no duplicate replay of already-delivered events? This is what `FromStream.After(...)` gives today. +4. **Filter drift / recreation cost.** Attempt to change an existing index's filter in place. Confirm whether it's + truly impossible (per the doc comment) or just unsupported by the current HTTP surface `UserDefinedIndexSource` + uses — and if impossible, measure delete+recreate+backfill latency for a realistic read-model event-type-set + size, since that becomes the "projection update" cost equivalent. +5. **Body-property routing (shape B).** Confirm directly (don't infer from the filter grammar) whether any + KurrentDB 26.1 index filter or fields configuration can express "route event to a target keyed by + `e.body.RecipientId`" — i.e., something with GROUP-BY/redirect semantics, not just a WHERE predicate. Expected + FAIL per §1/§2(a) reasoning above, but the spike should try it, not assume it. +6. **Index count ceiling.** Create O(50-100) indexes (proxy for read-model count if every `[OutputStream]` model + migrates) and confirm KurrentDB doesn't degrade/reject beyond some N — relevant to whether index-per-read-model + is operationally sane at all. + +## 4. Proposed integration design (revised post-spike — see §7 for the evidence this rests on) + +- **Not a drop-in stream-name swap — the subscription primitive itself must change.** The spike's biggest + design-relevant finding: `Client.SubscribeToStream("$idx-user-", ...)` — the obvious first thing anyone + would try, and exactly what `SubscriptionRunnerState.Subscribe()` (`SubscriptionRunner.cs:25-30`) calls today + against ordinary output streams — silently returns zero events, no exception (§7, SPIKE-2a). The only working + mechanism is `Client.SubscribeToAll(FromAll.Start, resolveLinkTos:true, filterOptions: StreamFilter.Prefix("$idx-user-"))` + (§7, SPIKE-2b). So the index-backed path needs a **new** `SubscriptionRunnerState`-equivalent built around + `SubscribeToAll`+filter, not a parameter tweak on the existing one. Anyone extending this later must not + "simplify" it back to `SubscribeToStream` against the index name — that's the footgun, and it fails silently + rather than loudly, so it needs an explicit code comment / assertion, not just tribal knowledge. +- **Checkpoint type changes from stream-local to `$all Position`.** Today's catch-up resume + (`SubscriptionRunner.cs:177`, `subscription.Position = FromStream.After(e.OriginalEventNumber)`) and the + persistent-subscription server-side checkpoint are both keyed off stream-local revision numbers. §7 SPIKE-3 + confirms the index-backed resume key is a `Position` (`C:.../P:...`) from the underlying `$all` log, not a + revision on `$idx-user-` — resuming means `FromAll.After(lastPosition)`, not `FromStream.After(lastRevision)`. + Any persisted subscription checkpoint (wherever a consumer stores "where I got to") needs a type that can hold + either shape, or a distinct index-backed checkpoint representation — this is a breaking change to whatever + checkpoint persistence format exists today if it assumes a `StreamPosition`/`ulong`. +- **New entry point, not a swap of `TryCreateJoinProjection`'s internals.** Add an index-backed sibling next to + `SubscribeEventHandler`/`SubscribeEventHandlerPersistently` in `PlumberEngine` (e.g. + `SubscribeEventHandlerViaIndex`) that: (1) calls an `EnsureIndexAsync(outputStream, eventTypes)` + analogous to `TryCreateJoinProjection` but POSTing `/v2/indexes/` with `BuildFilter` instead of creating a + projection; (2) subscribes via the new `SubscribeToAll`+filter state above, dispatching through the **same** + `OnEvent`/`ICaughtUpHandler`/error-backoff plumbing (`SubscriptionRunner.OnEvent`, `Plumber.cs`) so only the + event *source* changes, not the consumption model. +- **Persistent-subscription consumers are a HARD, confirmed exclusion — not a scoping choice pending more + evidence.** `SubscribeEventHandlerPersistently` / `SubscriptionSet.SubscribePersistentlyAsync` — used today by + `ProcessManagerClient`'s Inbox/Outbox (a real shape-A merge in this repo) — rely on + `PersistentSubscriptionClient.SubscribeToStream(stream, group)`, server-side group checkpointing, competing + consumers. §7 SPIKE-7 proves `PersistentSubscriptionClient.CreateToAllAsync(group, StreamFilter.Prefix("$idx-user-"), ...)` + creates and connects with **no error** but delivers **zero index entries**, catch-up or live, under every + variant tried (`resolveLinkTos` true/false, `CreateToStream` directly on the index stream) — controlled against + a persistent filtered-`$all` subscription over an ordinary `EventTypeFilter` on the SAME node, which correctly + delivered 3/3 (ruling out a broken harness). **Persistent subscriptions genuinely cannot see user-defined-index + links.** This is not "unproven, wait for more data" — it's a measured FAIL with a sound control. Index-backed + merge is **catch-up-subscription-only, by design constraint, permanently** — `SubscribeEventHandlerPersistently` + and every handler that uses it (`ProcessManagerClient` included) stay on projections; there is no future path + to close this gap short of a KurrentDB server change outside this framework's control. +- **Coexistence, not replacement, initially.** Keep `ensureOutputStreamProjection` (rename conceptually to + "ensure merge source") as a discriminated choice — projection-backed (today, still required for + shape B and for persistent subscriptions until SPIKE-5) vs index-backed (new, catch-up-only, shape A) — + selected per handler, so existing apps keep working unchanged and only opt-in handlers move. This mirrors how + `MigrationRunner` already treats `ProjectionCopyContext`/`UserDefinedIndexCopyContext` as mutually exclusive but + coexisting strategies (`UserDefinedIndexCopyContext.cs:27-28`). +- **Change-detection equivalent to `mp_query_hash`.** Indexes can't be redefined in place (per the code comment + in `UserDefinedIndexSource.cs`; not independently re-verified by the spike, which only ever created fresh + indexes). The safest analog is: hash the filter, store it as custom metadata *on the read stream side* (there's + no index-native metadata slot), and on mismatch either (a) fail loud requiring an operator-driven + delete+recreate+rebuild, or (b) create a *new* suffixed index and cut consumers over — both are heavier than + today's disable/update/enable. Latency of delete+recreate+backfill for a realistic event-type-set size is still + unmeasured (§3 item 4 was not part of this spike round). +- **Shape (B) IS index-backable — reversed from the first spike round, see §8.** §7 SPIKE-4's "no index + equivalent" verdict rested on the wrong body accessor and is superseded. §8 SPIKE-8 shows the correct recipe: + create the index with a single `"fields"` entry (`{"name":"recipientid","selector":"rec => rec.value.RecipientId","type":"INDEX_FIELD_TYPE_STRING"}`) + — this is the `EnsureIndexAsync`-analog for `EnsureLookupProjection`, call it `EnsureFieldPartitionedIndexAsync(category, eventProperty, outputStreamCategory)`. + Reading a specific key's partition is `SubscribeToAll(..., filterOptions: StreamFilter.Prefix("$idx-user-:"))` + — same subscription primitive as shape (A) (§8 SPIKE-9 reconfirms direct `SubscribeToStream`/`ReadStreamAsync` + against a partition name fails the same way as the whole-index stream: `StreamNotFound`/zero-events-no-error), + same `$all`-`Position` checkpoint discipline, same persistent-subscription exclusion (§7 SPIKE-7's finding is a + subscription-mechanism limitation, not a filter limitation, so it applies identically to partition streams — + not independently re-spiked against a partition name specifically, but the same underlying "index links live + in `$all` only" mechanism makes a different outcome very unlikely). **Translation gap — CLOSED, see §8 SPIKE-12**: + `EnsureLookupProjection`'s source is `fromStreams(['$ce-{category}'])` — *every* event in a stream **category**, + regardless of type — and the index filter CAN express this directly: `rec.position.stream.startsWith("-")`, + proven with a control that rules out type-based matching as the explanation (§8 SPIKE-12). So + `EnsureFieldPartitionedIndexAsync`'s source-selection is a **direct category filter**, not an enumerated + OR-of-known-event-types — no "misses a type added later" caveat. The one-field-per-index cap (§2(a)) still + means this only covers a single routing property, matching `EnsureLookupProjection`'s one `eventProperty` + parameter exactly. +- **`ProcessManagerClient`'s Lookup vs. Inbox/Outbox split.** Per §1's correction: the repo's one shape-B call + site (`{ProcessManager}Lookup`, read via `Rehydrate` — bounded, catch-up-only) is now a real index-backing + candidate. Its sibling streams from the SAME call (`Inbox`/`Outbox`, shape A, consumed via + `SubscribeEventHandlerPersistently`) are NOT, per the persistent-subscription exclusion above. A migration + would move `Lookup` and leave `Inbox`/`Outbox` on projections — a within-one-feature split, not an all-or-nothing + cutover for `ProcessManagerClient`. + +## 5. Risk register (updated post-spike; RESOLVED rows cite §7 evidence) + +| # | Risk | Where it bites | Status | +|---|------|-----------------|--------| +| 1 | Live catch-up+tail subscription primitive for `$idx-user-*` | Every read model that would move | **RESOLVED — exists**, but only via `SubscribeToAll`+`StreamFilter.Prefix` (§7 SPIKE-2b); direct `SubscribeToStream` on the index stream does NOT work (§7 SPIKE-2a, see risk #9) | +| 2 | Commit order / liveness under concurrent live appends (not just offline backfill) | Same | **RESOLVED — PASS** (§7 SPIKE-1: 10 interleaved events across a backfill+live-append boundary read back in exact commit order) | +| 3 | Resumable position semantics | Process restart of a live consumer | **RESOLVED — PASS, but the checkpoint TYPE changes**: resume key is `$all Position`, not a stream revision (§7 SPIKE-3) — see §4 design note, this is a breaking change to any `StreamPosition`-typed checkpoint store | +| 4 | Index cannot be redefined in place | Every handler code change that adds/removes an event type — today cheap disable/update/enable; would become delete+rebuild | **RESOLVED — measured** (§7 SPIKE-5): in-place mutation is REJECTED, not silently swapped — re-POSTing an existing name with a different filter returns HTTP 409 and `GET` confirms the filter is UNCHANGED. `DELETE /v2/indexes/` IS supported (HTTP 200). Cost = delete+recreate+full-backfill; measured **~3.7s end-to-end for 10,000 matching events**. So the `mp_query_hash` equivalent is "delete, recreate under a reusable name, wait for full re-backfill" — cheap at 10k events, but genuinely a full re-backfill each change, not an incremental redefine; cost scales with matched-event count, which matters for high-volume event types | +| 5 | `$ce` + body-property fan-out (shape B) has no expressible index filter equivalent | `ProcessManagerClient` lookup streams | **SUPERSEDED — REOPENED THEN RESOLVED AS PASS** (§7 SPIKE-4's FAIL used the wrong body accessor; §8 SPIKE-8 shows `rec.value.*` + a `"fields"` partition config materializes exactly this pattern as `$idx-user-:` partition streams, catch-up-subscription-only, one routing property max) | +| 6 | Custom `linkTo` transforms in general (anything beyond "copy the event verbatim to a named target") have no index equivalent | Any future handler using a computed `linkTo` target, not just the current lookup case | **Narrowed further, still open**: the single-property key-routing case (what `EnsureLookupProjection` actually does, INCLUDING its `$ce`-category source) IS fully covered now (see #5, #11). What remains genuinely unaddressed: multi-property/composite routing keys (blocked by the one-field-per-index cap) and any transform beyond a straight body-field lookup (e.g. computed/derived values) | +| 7 | Index proliferation ceiling unknown | 20 `[OutputStream]` handlers today, growing; naive 1:1 index-per-model | **RESOLVED for 3x current scale** (§7 SPIKE-6): 60 indexes created on one node in 1268ms, 0 rejections; first- and last-created (`idx-000`, `idx-059`) both fully backfilled 100 events in commit order under concurrent backfill load — no degradation or starvation observed. Not pushed past 60 (3x today's 20 read models); no evidence of a hard ceiling but the true limit (100s-1000s?) is still unknown. Note this is whole-index count — a *partitioned* index (shape B) stays at one index regardless of key cardinality (§2(d) correction), so this risk is really about shape-A's 1:1 model count, not about lookup-key count | +| 8 | Two parallel merge mechanisms (projection- and index-backed) increase operational surface if coexistence becomes permanent | Ops/debugging | Open — design-scope decision, not empirical | +| 9 | `SubscribeToStream` directly against `$idx-user-` (or a partition `$idx-user-:`) silently succeeds-but-delivers-nothing (no exception, `caughtUp=false`, zero events) | Anyone porting the existing `SubscriptionRunnerState.Subscribe()` pattern naively to an index or partition stream gets a silent hang with no diagnostic | **High** — reconfirmed for the whole-index case by §8 SPIKE-9 (`ReadStreamAsync` → `StreamNotFound`, `SubscribeToStream` → 0 events no error, both `resolveLinkTos` settings); not independently re-tested against a partition name but the same underlying mechanism makes the same failure near-certain. This is a footgun, not a blocker (the working mechanism exists), but it WILL bite a future implementer who doesn't read this doc; needs a guard/comment at the call site, ideally a debug-time assertion that filtered `$all` is used instead of raw stream-subscribe for any `$idx-user-` prefixed name | +| 10 | Persistent subscriptions (server-checkpointed, competing consumers) cannot see user-defined-index links | `ProcessManagerClient`'s Inbox/Outbox (`SubscribeEventHandlerPersistently`) — a real shape-A merge already using persistent subscriptions today; ALSO applies to a would-be shape-B persistent consumer | **RESOLVED — CONFIRMED FAIL, permanent hard boundary, applies to BOTH shapes** (§7 SPIKE-7): `CreateToAllAsync` + `StreamFilter.Prefix("$idx-user-")` connects with no error but delivers ZERO entries under every variant tried; a same-node control (persistent filtered-`$all` over an ordinary `EventTypeFilter`) delivered 3/3, ruling out a broken harness. Index-backed merge is catch-up-subscription-only, by design constraint, for shape A AND shape B alike; `SubscribeEventHandlerPersistently` handlers stay on projections with no future closure path inside this framework | +| 11 | `EnsureLookupProjection`'s `$ce-{category}` source (every event in a stream category) — is there a proven index filter equivalent, or only event-**type** predicates (`rec.schema.name`)? | Any `EnsureFieldPartitionedIndexAsync` built from this framing | **RESOLVED — PASS** (§8 SPIKE-12): `rec.position.stream.startsWith("-")` (or `.split('-')[0] == ""`) selects by stream/category directly, proven against a control (same event type in two categories; type-based control filter matches both, category filter matches only one) — no OR-of-event-types workaround needed, no "misses a type added later" caveat | + +## 6. What definitively CANNOT be replaced by a plain event-type-filtered index + +- **Multi-property / composite routing keys** — the `"fields"` config that makes shape (B) work (§8 SPIKE-8) is + capped at **one field per index** (KurrentDB docs); a routing key needing more than one body property has no + index equivalent. `EnsureLookupProjection`'s single-`eventProperty` design fits within the cap today, so this + is a forward-looking limit, not a current blocker. +- **Persistent-subscription consumption, for either shape** — §7 SPIKE-7: persistent subscriptions cannot see + index links at all, confirmed with a sound control. This is the ONLY boundary that survived all three spike + rounds unchanged and un-superseded — the sole permanent, hard exclusion this investigation found. +- Any hypothetical future `linkTo` with a genuinely *transformed* event body (not just a verbatim copy routed by + a single field) — the index selects/partitions original records, it does not compute new derived content. +- Direct `SubscribeToStream`/`ReadStreamAsync` against `$idx-user-` or a partition name — not a capability + gap in the index itself, but a client API that looks like it should work and doesn't (risk #9); the actual read + path is `SubscribeToAll`+filter for both the whole index and its partitions. + +**No longer on this list (moved to "CAN be replaced"): shape (B) single-property body-routing lookups +(`EnsureLookupProjection` as actually used by `ProcessManagerClient` today, INCLUDING its `$ce`-category source +selection, §8 SPIKE-8/12), for catch-up consumers.** + +## 7. Empirical spike results (`eng-mergefeas`, KurrentDB 26.1.0.3443, MicroPlumberd.Testing Docker) + +Raw evidence in `src/MicroPlumberd.Migration.Tests/LiveIndexSubscriptionSpikes.cs` (uncommitted, clearly labelled +SPIKE — not shipped code). + +- **SPIKE-1 (live tail ordering) — PASS.** Index over types A,B built on `A0,B1,A2,B3,A4` → read back `0,1,2,3,4`. + Then `B5,A6,B7,A8,B9` appended to a stream AFTER the index existed → full read back `0..9` in exact commit + order. Confirms live growth stays commit-ordered, not just the backfill case already proven in + `dev-log-userdefined-index.md`. +- **SPIKE-2 (subscribe: catch-up → live) — PASS overall, via one mechanism, with a critical negative on the other.** + - 2b `SubscribeToAll(FromAll.Start, resolveLinkTos:true, filterOptions: StreamFilter.Prefix("$idx-user-"))` + — PASS. One open subscription caught up `0,1,2`, then received live-appended `3,4` pushed with no re-read, + commit order preserved. This is the load-bearing mechanism. + - 2a `SubscribeToStream("$idx-user-", FromStream.Start, resolveLinkTos:true)` — FAIL, silently: no + exception, `caughtUp=false`, zero events delivered. The index's links live in `$all` only; the + `$idx-user-` name is not itself a directly-subscribable stream. Logged as risk #9. +- **SPIKE-3 (checkpoint/resume) — PASS.** Recorded the last-seen link's `$all` `Position` (`e.OriginalPosition`, + e.g. `C:11302/P:11302`) during a first pass (`0,1,2`), stopped, appended `3,4`, resumed a NEW + `SubscribeToAll(FromAll.After(checkpoint), ...)` → received only `3,4`, no replay of `0,1,2`. Resume key is an + `$all Position`, not an index-stream revision — a type change from today's `FromStream`/`StreamPosition`-based + checkpoints (§4). +- **SPIKE-4 (body-property routing, shape B) — FAIL, confirmed with a valid control.** First attempt showed a 400 + that turned out to be an index-NAME validation error, not a filter rejection — re-run with a fixed name. + Corrected run: `rec.data.SomeId=="x"` / `rec.body.SomeId=="x"` / `rec.data.someid=="x"` filters are all ACCEPTED + (HTTP 200) but the index indexes ZERO events where a body/data-partitioned split should have produced 2 of 4. + Control filter `rec.schema.name=="UdiC"` (same lowercase-name scheme, same data) correctly indexed all 4 — so + the empty result isn't a naming/timing artifact, the filter genuinely cannot see event body content. Plus the + structural limit stands: an index is one filtered stream, it cannot fan out into N per-value streams the way + `linkTo('cat-' + e.body.SomeId)` does. +- **SPIKE-5 (filter drift / recreation cost) — measured.** Re-`POST`ing an existing index name with a *different* + filter returns HTTP 409 `INDEX_ALREADY_EXISTS`, and a follow-up `GET` confirms the filter is left UNCHANGED — + no silent redefinition, matching the code-comment's claim (now independently verified, not just trusted). + `DELETE /v2/indexes/` IS supported (HTTP 200). Full delete → recreate (same name) → backfill of 10,000 + matching events measured at **~3.7s (3655ms) end-to-end to ready**. Confirms risk #4's `mp_query_hash` + equivalent: change-detection still means delete+recreate+full-rebuild, cost scaling with matched-event count, + not an incremental filter update. +- **SPIKE-6 (index count ceiling) — PASS at 3x current scale.** Created 60 indexes on a single node in 1268ms + total, zero rejections. Both the first-created (`idx-000`) and last-created (`idx-059`) fully backfilled their + 100 events in commit order despite concurrent backfill load across all 60 — no degradation, no starvation of + early vs. late indexes. Not pushed to a hard ceiling (60 vs. today's 20 `[OutputStream]` handlers); `eng-mergefeas` + offered to extend to 200 if a harder number is wanted before committing to 1-index-per-read-model at scale. +- **SPIKE-7 (persistent subscription to filtered `$all`) — CONFIRMED FAIL, permanent hard boundary.** + `PersistentSubscriptionClient.CreateToAllAsync(group, StreamFilter.Prefix("$idx-user-"), settings{resolveLinkTos:true, startFrom:Position.Start})` + creates and connects with **no error** but delivers **zero index entries**, catch-up or live. Four variants on + the same node/index ruled out every alternate explanation: (A) persistent filtered-`$all` over an ordinary + `EventTypeFilter` on domain events → 3/3 delivered — proves the harness itself is sound; (B) `$idx-user` prefix, + `resolveLinkTos:false` → 0; (C) same prefix, `resolveLinkTos:true` → 0; (D) `CreateToStream` directly on the + `$idx-user-` stream (persistent, not filtered-`$all`) → 0, no error. A=3 vs. B/C/D=0 on the same index + is conclusive: **persistent subscriptions genuinely cannot see user-defined-index links**, by any mechanism + tried. The catch-up path (SPIKE-2b) sees them fine — the split between catch-up and persistent is clean and + total. `eng-mergefeas` offered to further probe whether persistent subscriptions fault loudly at scale or stay + silently empty; not pursued — the boundary for the design is already unambiguous without it. + +**One-line empirical verdict as of §7 (eng-mergefeas, since superseded on shape B by §8 below):** a live, +commit-ordered, subscribable index-backed merge stream IS achievable on KurrentDB 26.1, but ONLY via filtered +`SubscribeToAll` (catch-up, client-checkpointed with an `$all`-`Position`) over `$idx-user-`, and ONLY for +event-type/stream merges (shape A). Body-property lookup projections (shape B) and anything using persistent +subscriptions must stay as projections. **§8 reopens and reverses the shape-B half of this** — kept here verbatim +as the historical record of what was believed after round one; do not treat this paragraph as current without +reading §8. + +## 8. Second empirical batch (`eng-mergefeas`, same environment) — shape-B reversal + direct-subscribe reconfirmation + +Raw evidence in `src/MicroPlumberd.Migration.Tests/FieldPartitionAndDirectSubscribeSpikes.cs` (uncommitted, +SPIKE-8/9/10, all green). + +- **SPIKE-8 (field-partitioned index for body-property routing) — supersedes SPIKE-4's FAIL.** SPIKE-4 used the + wrong body accessor (`rec.data`/`rec.body`); the correct one is `rec.value`. More importantly, KurrentDB 26.1 + has a dedicated `"fields"` index config, orthogonal to the filter: + `POST /v2/indexes/ {"filter":"rec => rec.schema.name=='UdiC'", "fields":[{"name":"someid","selector":"rec => rec.value.SomeId","type":"INDEX_FIELD_TYPE_STRING"}], "start":true}` + (max **one** field per index, per the KurrentDB docs). This materializes a **partition stream per distinct + field value** — `$idx-user-:`. Test: interleaved `alpha=0,2,4` / `beta=1,3` events → reading + `$idx-user-:alpha` via `StreamFilter.Prefix` returns exactly `0,2,4` (only that key, correct commit + order); `:beta` returns `1,3`. Live-appending `Ord6` tagged `alpha` → `:alpha` becomes `0,2,4,6`, live and + ordered. This is the index-native analog of `EnsureLookupProjection`'s `linkTo('cat-' + e.body.RecipientId, e)` + — a field-keyed partitioned index, consumed per-partition, replaces a per-key lookup projection **for catch-up + consumers**. The SPIKE-7 persistent-subscription exclusion is unaffected — it's the same underlying + index-links-live-in-`$all`-only mechanism, so it applies to partitions too (not independently re-run against a + partition name, but there's no reason to expect a different result). +- **SPIKE-9 (direct-subscribe, definitive) — reconfirms and generalizes risk #9.** For the *whole* index stream + (not yet independently tested per-partition): `ReadStreamAsync` → `StreamNotFound` (both `resolveLinkTos` + settings); `SubscribeToStream` → 0 events, no error (both settings). The index's content is **never + materialized as an ordinary directly-subscribable/readable stream** — only the filtered `$all` path + (`SubscribeToAll`/`ReadAllAsync` + `StreamFilter.Prefix`) sees it, for the whole index and (per SPIKE-8) for + partitions alike. This is the conclusive version of what SPIKE-2a first observed: any new `SubscriptionRunnerState` + variant for index-backed merge **must** be built on `Client.SubscribeToAll(FromAll, resolveLinkTos:true, SubscriptionFilterOptions(StreamFilter.Prefix(...)))` + — a stream-name swap on the existing `Client.SubscribeToStream` path silently returns nothing, with no + exception to catch the mistake. +- **SPIKE-10 (`CaughtUp` signal + no-count-gate on the live path) — confirms two design details for §4.** + `StreamMessage.CaughtUp` fires correctly at the history→live boundary (observed after 3 historical events, + immediately following history, before any live append) — so `ICaughtUpHandler.CaughtUp()` wiring (§2(b)) works + unmodified against the index-backed subscription. Separately: the live-tail path must **not** use + `UserDefinedIndexSource.WaitUntilReadyAsync`'s exact-count gate — that gate is correct for the *offline + migration* case (a frozen, known total) but would **hang forever** against a live, growing store where the + "expected count" is a moving target. The live read-model path fires ready on `CaughtUp`, not on a count match. +- **SPIKE-11 (per-key field-partition SUBSCRIBE, not just read) — closes the last gap in the shape-B reversal.** + §8 SPIKE-8 proved partitions are *readable* per-key in commit order; SPIKE-11 proves the same for the live + *subscribe* path, with clean key isolation: `SubscribeToAll(FromAll.Start, resolveLinkTos:true, SubscriptionFilterOptions(StreamFilter.Prefix("$idx-user-:alpha")))` + catches up on ONLY that key's history (`0,2,4`, `CaughtUp` fires), then, after `beta`'s key gets a live append + (`5`) followed by `alpha`'s (`6`), the open subscription receives **only** `6` — no `beta` leak, final sequence + `0,2,4,6`, commit order preserved. So a per-key lookup consumer (the `ProcessManagerClient.Lookup`-style read) + is a genuine live catch-up subscription with the SAME recipe as the whole-index case, just a longer + `StreamFilter.Prefix` (`"$idx-user-:"` instead of `"$idx-user-"`) — no separate mechanism, + no cross-key bleed to guard against in the consumption code. +- **SPIKE-12 (`$ce`-category filter expressivity, risk #11) — PASS, closes the last open question.** The + KurrentDB 26.1 `rec` surface (per the docs page cited): `rec.schema.name` = event type, `rec.value` = body, + `rec.properties` = metadata, `rec.position.stream`/`streamRevision`/`logPosition`, `rec.id`/`timestamp`/`sequence` + — no dedicated "category" property, so category = the prefix before the first `-` in `rec.position.stream`, same + convention MicroPlumberd itself uses. Test deliberately makes category the ONLY discriminator: event type `"Ev"` + appears in BOTH category X (streams `X-1`,`X-2` → `Ord 0,2`) and category Y (`Y-1`,`Y-2` → `Ord 1,3`). Control + `rec => rec.schema.name == 'Ev'` → indexed all four (`0,1,2,3`) — proves a type-only filter can't distinguish X + from Y, i.e. the harness is sound and category selection is a genuinely different capability. Two category-filter + forms — `rec => rec.position.stream.startsWith('X-')` and `rec => rec.position.stream.split('-')[0] == 'X'` — + both indexed ONLY `0,2` (category X alone). Control-matches-all vs. category-filter-matches-only-X is conclusive: + **stream/category-based selection is directly expressible**, no event-type enumeration needed. This closes §4's + translation gap: `EnsureFieldPartitionedIndexAsync`'s source-selection filter is + `rec => rec.position.stream.startsWith("-")`, a direct translation of `fromStreams(['$ce-{category}'])`, + not a workaround. + +**Standardized recipe (eng-mergefeas, citing KurrentDB's `features/indexes/user-defined.html`), covering both +shapes — event-type OR category-based source selection, whole-index or per-key-partition reads — for catch-up +consumers:** +`SubscribeToAll(FromAll.Start, resolveLinkTos:true, SubscriptionFilterOptions(StreamFilter.Prefix("$idx-user-"[+":"+value])))`, +checkpoint on `ResolvedEvent.OriginalPosition` (an `$all Position`), resume via `FromAll.After(pos)`, fire ready +on `CaughtUp` (no count-gate on the live path — that's migration-only); source filter is +`rec.schema.name == "..."` (shape A, event-type join) or `rec.position.stream.startsWith("-")` +(shape B's `$ce`-category source), optionally combined with a `"fields"` selector for per-key partitioning. + +## Coordination — closed, no spikes outstanding + +This framing (§1-§2) fed `eng-mergefeas`'s empirical spike (§3); all six original §3 questions, the +persistent-subscription follow-up, the shape-B reversal, the per-key subscribe confirmation, and the `$ce`-category +filter question all now have empirical answers (§7 SPIKE-1..7, §8 SPIKE-8..12), folded into §1/§2/§4/§5/§6 above. +Every risk-register row that could be resolved empirically (#1, #2, #3, #4, #5, #7, #9, #10, #11) is now RESOLVED; +the only rows that remain open are design-scope judgment calls with no empirical answer to seek — #6 (narrowed to +composite-key/transform cases the one-field cap already rules out) and #8 (dual-mechanism operational surface). +**No further spike is outstanding or anticipated. This feasibility investigation is complete, across three spike +rounds and one clean reversal, fully folded in.** + +The bounded, confirmed scope for an implementation design: index-backed merge covers **catch-up-subscription** +consumers for BOTH shape (A) (event-type/stream merges) and shape (B) (single-property body-routing lookups, via +field-partitioned indexes, including a `$ce`-category source translated directly, not enumerated). It permanently +excludes anything using `SubscribeEventHandlerPersistently` — for either shape — which stays projection-backed. +In this repo concretely: the 20 `[OutputStream]` catch-up handlers and `ProcessManagerClient`'s `Lookup` stream +are index-backing candidates; `ProcessManagerClient`'s `Inbox`/`Outbox` +(persistent) are not. Handed off to task M1 / `design-mergeidx` +for `design.md`. diff --git a/src/MicroPlumberd.Migration.Runner/MicroPlumberd.Migration.Runner.csproj b/src/MicroPlumberd.Migration.Runner/MicroPlumberd.Migration.Runner.csproj new file mode 100644 index 0000000..8a76cef --- /dev/null +++ b/src/MicroPlumberd.Migration.Runner/MicroPlumberd.Migration.Runner.csproj @@ -0,0 +1,25 @@ + + + + Exe + net10.0 + enable + enable + false + mp-migrate + MicroPlumberd.Migration.Runner + + + + + + + + + + + + + + + diff --git a/src/MicroPlumberd.Migration.Runner/Migrations/_0001_DropTombstonedTestOffers.cs b/src/MicroPlumberd.Migration.Runner/Migrations/_0001_DropTombstonedTestOffers.cs new file mode 100644 index 0000000..7842fda --- /dev/null +++ b/src/MicroPlumberd.Migration.Runner/Migrations/_0001_DropTombstonedTestOffers.cs @@ -0,0 +1,17 @@ +namespace MicroPlumberd.Migration.Runner.Migrations; + +/// +/// Drops the five hard-tombstoned test Offer streams left on the :5081 store. Their events must not be +/// carried into the fresh destination; the copy engine additionally skips the tombstone markers themselves. +/// +public sealed class _0001_DropTombstonedTestOffers : Migration +{ + public override string Id => "0001_drop_tombstoned_test_offers"; + + public override void Migrate(IMigrationBuilder b) => b.DropStreams( + "Offer-of-ee959052-0d0e-481e-b45f-c3ce7a6007a7", + "Offer-of-3d619f92-976e-4201-8510-ce41123d5395", + "Offer-of-b8eeae43-078c-47f5-b262-84c686fd44fd", + "Offer-of-f098b22e-e55f-4f46-92bf-1869e1cb7221", + "Offer-of-d5b2d31f-5a5c-4561-9074-7b97ef50f64a"); +} diff --git a/src/MicroPlumberd.Migration.Runner/Program.cs b/src/MicroPlumberd.Migration.Runner/Program.cs new file mode 100644 index 0000000..b6c8c01 --- /dev/null +++ b/src/MicroPlumberd.Migration.Runner/Program.cs @@ -0,0 +1,111 @@ +using System.Reflection; +using KurrentDB.Client; +using Microsoft.Extensions.Logging; +using MicroPlumberd.Migration; + +// mp-migrate --source --dest [--dry-run] +// +// Offline event-store rewrite. SOURCE is read-only; DEST must be a fresh, empty store (topology is +// provided externally — e.g. source ES on the existing volume RO + dest ES on a fresh volume, both up). + +var args0 = ParseArgs(args); +if (!args0.TryGetValue("source", out var sourceConn) || !args0.TryGetValue("dest", out var destConn)) +{ + Console.Error.WriteLine("Usage: mp-migrate --source --dest [--dry-run]"); + Console.Error.WriteLine(" --source SOURCE KurrentDB connection string (read-only)."); + Console.Error.WriteLine(" --dest DEST KurrentDB connection string (fresh, empty store)."); + Console.Error.WriteLine(" --dry-run Report per-migration counts; write nothing."); + Console.Error.WriteLine(); + Console.Error.WriteLine("REQUIREMENT: the DEST EventStore MUST run with standard projections enabled"); + Console.Error.WriteLine("($by_event_type running). The tool PRE-CREATES the app's [OutputStream] join projections"); + Console.Error.WriteLine("on the fresh dest and PACES the event copy (per-event link emission) so those merge"); + Console.Error.WriteLine("streams are rebuilt in commit/arrival order; the app NO-OPs on boot (mp_query_hash match)."); + Console.Error.WriteLine("If standard projections are off, $et never repopulates and the join/output streams stay empty."); + return 2; +} + +var dryRun = args0.ContainsKey("dry-run"); + +using var loggerFactory = LoggerFactory.Create(b => b + .SetMinimumLevel(LogLevel.Information) + .AddSimpleConsole(o => { o.SingleLine = true; o.TimestampFormat = "HH:mm:ss "; })); +var log = loggerFactory.CreateLogger("mp-migrate"); + +var sourceSettings = KurrentDBClientSettings.Create(sourceConn); +var destSettings = KurrentDBClientSettings.Create(destConn); +await using var source = new KurrentDBClient(sourceSettings); +await using var dest = new KurrentDBClient(destSettings); +var sourceProjections = new KurrentDBProjectionManagementClient(sourceSettings); +var destProjections = new KurrentDBProjectionManagementClient(destSettings); + +// PACED PROJECTION COPY (default on a real migration): pre-create the source's [OutputStream] join projections +// on the fresh dest and pace the copy per-event so each merge stream is rebuilt in commit order. On --dry-run +// the runner respects dryRun end-to-end (pre-creates nothing, no pacing) and reports what WOULD be created. +var projectionCopy = new ProjectionCopyContext +{ + SourceProjections = sourceProjections, + DestProjections = destProjections, + SourceConnectionString = sourceConn +}; + +var migrations = MigrationDiscovery.FromAssemblies(Assembly.GetExecutingAssembly()); +log.LogInformation("Discovered {Count} migration(s): {Ids}", migrations.Count, + string.Join(", ", migrations.Select(m => m.Id))); +log.LogWarning("DEST must run with standard projections enabled ($by_event_type). The tool pre-creates the app's " + + "[OutputStream] join projections on the dest and PACES the copy (per-event link emission) so the " + + "merge streams rebuild in commit order; with projections off they stay empty."); + +try +{ + var result = await new MigrationRunner(loggerFactory) + .RunAsync(source, dest, migrations, dryRun, projectionCopy); + + Console.WriteLine(); + Console.WriteLine(dryRun ? "=== DRY RUN (nothing written) ===" : "=== MIGRATION COMPLETE ==="); + Console.WriteLine($"Pending applied: {(result.PendingMigrationIds.Count == 0 ? "(none)" : string.Join(", ", result.PendingMigrationIds))}"); + Console.WriteLine($"Source events scanned: {result.Copy.SourceEvents}"); + Console.WriteLine($"Kept: {result.Copy.Kept} Dropped: {result.Copy.Dropped}"); + Console.WriteLine($"Merge/link events skipped (source $> links, rebuilt by the paced dest projections): {result.Copy.LinkEventsSkipped}"); + if (result.CopiedProjections.Count > 0) + Console.WriteLine($"Projections {(dryRun ? "that WOULD be pre-created" : "pre-created + paced on the dest")}: {string.Join(", ", result.CopiedProjections)}"); + Console.WriteLine("Per-migration:"); + foreach (var id in result.PendingMigrationIds) + { + var s = result.Copy.MigrationStats[id]; + Console.WriteLine($" {id}: dropped={s.Dropped}, renamed={s.Renamed}, transformed={s.Transformed}"); + } + + // Unparseable-but-declared-JSON payloads are copied VERBATIM (byte-for-byte), never dropped — a benign + // warning, not data loss, so it does NOT fail the run. + if (result.Copy.UnparseableVerbatim > 0) + Console.WriteLine($"NOTE: {result.Copy.UnparseableVerbatim} source event(s) had unparseable JSON payloads " + + "and were copied VERBATIM (byte-for-byte). No data loss."); + + if (result.Verification is not null) + { + Console.WriteLine(); + Console.WriteLine(result.Verification.Format()); + return result.Verification.AllOk ? 0 : 1; + } + return 0; +} +catch (MigrationChecksumMismatchException ex) +{ + log.LogError(ex, "Refusing to run: an already-applied migration changed."); + return 3; +} + +static Dictionary ParseArgs(string[] argv) +{ + var d = new Dictionary(StringComparer.OrdinalIgnoreCase); + for (var i = 0; i < argv.Length; i++) + { + if (!argv[i].StartsWith("--", StringComparison.Ordinal)) continue; + var key = argv[i][2..]; + if (i + 1 < argv.Length && !argv[i + 1].StartsWith("--", StringComparison.Ordinal)) + d[key] = argv[++i]; + else + d[key] = "true"; // flag + } + return d; +} diff --git a/src/MicroPlumberd.Migration.Runner/RUNBOOK.md b/src/MicroPlumberd.Migration.Runner/RUNBOOK.md new file mode 100644 index 0000000..d066e5b --- /dev/null +++ b/src/MicroPlumberd.Migration.Runner/RUNBOOK.md @@ -0,0 +1,143 @@ +# mp-migrate — offline event-store rewrite runbook + +`mp-migrate` rewrites a SOURCE KurrentDB store into a fresh DEST store, applying raw (non-typed) +migration rules. It is offline: point it at two already-running stores. + +``` +mp-migrate --source --dest [--dry-run] +``` + +- `--source` — SOURCE connection string (read-only). +- `--dest` — DEST connection string. Must be a **fresh, empty** store. +- `--dry-run` — report per-migration Kept/Dropped/Renamed/Transformed counts; write nothing. + +## Topology + +Provide the topology externally (e.g. docker): source ES on the existing volume mounted read-only, +dest ES on a fresh volume, both up. The tool only needs the two connection strings. + +## OPERATIONAL REQUIREMENT — standard projections MUST be enabled on DEST + +The DEST EventStore **must run with standard projections enabled and running** — specifically +`$by_event_type` (and `$by_category`). For a KurrentDB container: + +``` +KURRENTDB_RUN_PROJECTIONS=All +KURRENTDB_START_STANDARD_PROJECTIONS=true +``` + +Why: `[OutputStream]` merge/read-model streams are **not copied**. They are built by continuous +`linkTo` JOIN projections that the application re-registers on boot +(`SubscribeEventHandler(ensureOutputStreamProjection=true)`), reading from `$et-{EventType}`. The +migration tool deliberately **SKIPS** every `$>` link event (and `$ce`/`$et`/system streams) — a +count is reported as `Merge/link events skipped`. On the dest, the app's projection regenerates each +output stream from `$et`, in order, with correct dest links and no dead links (the dropped/tombstoned +aggregate stream is gone). If standard projections are OFF, `$et` never repopulates, so the +regenerated join/output streams stay **empty** and read models project nothing. + +Do **not** try to rebuild the merge streams during migration: the app's re-registered projection would +emit the same links a second time = duplicate links = corruption. Skip-and-regenerate is the design. + +## What the tool does + +- Reads SOURCE `$all` in commit order (`resolveLinkTos:false`, so tombstone/deleted-stream artifacts + are never dereferenced — no NRE). +- Three stream classes: `$`-system streams SKIPPED; `[OutputStream]` merge/link (`$>`) streams + SKIPPED + counted; aggregate/domain streams copied raw with per-stream expected version (gapless), + rules applied (Drop / Rename type / Transform JSON / Rename stream). Appends are BATCHED and capped by + both event COUNT and cumulative BYTES so no single append exceeds KurrentDB's limits; a stream larger than + one batch is flushed across several appends (gapless version preserved). +- Copies each event's metadata verbatim (correlation/causation/Created preserved) and stamps + `MigratedAt` (the run's UTC time). The **original event id is preserved** (each event is written once to a + fresh dest, so keeping the id holds causation/correlation-by-id refs + the idempotency key; `linkTo` uses + stream+position, not id, so read-model rebuild is unaffected). +- An event whose payload is **declared JSON but unparseable is NEVER dropped** — it is copied VERBATIM + (byte-for-byte) and reported as `UnparseableVerbatim` (a warning, not data loss). Only an explicit rule + (DropStream/DropStreams/DropEvent) drops an event. +- History travels in the `mp-migrations` stream (one `MigrationApplied` record per applied migration). + On each run it carries existing history forward, computes PENDING = defined-but-not-applied, and + refuses to run if an already-applied migration's checksum changed. +- Prints a verification report: per destination stream, expected vs actual counts, final versions, AND a + **write-fidelity checksum** — a per-stream SHA-256 over each written event's `(EventType || 0x00 || Data)` + recomputed by re-reading the dest, which catches dest-side reorder/truncation/corruption a count check + cannot. It verifies the copy INTENT reached the dest faithfully; it does NOT re-derive from the source or + re-check transform semantics (tests cover those). Any mismatch not explained by an intended drop is flagged. + +## Migration authoring notes + +- **Prefer literal arguments** (`DropStream("X")`, `DropStreams(...)`, `RenameType`, `RenameStream`) for + anything whose applied-history checksum must stay stable. Their checksum is over the literal text and is + build-independent. `_0001` is literal-only. +- **Delegate-based ops** (`DropStream(predicate)`, `DropEvent`, `TransformJson`) are checksummed over the + delegate's IL body. Two caveats: (a) IL can differ across build configuration / compiler / TFM, so an + already-applied delegate migration can trip the checksum guard on a later run under a different build — + build/run delegate-based migrations with a **pinned configuration**; (b) **the checksum does NOT see values + captured by the lambda** (e.g. a captured constant/list) — it hashes the delegate's IL body only, so two + migrations differing ONLY in a captured value hash the SAME and the guard will NOT catch such an edit (m7). + Keep delegate migrations self-contained (inline literals, no captured state) and avoid editing them after + they are applied. +- **Id ordering.** Migrations are applied in ORDINAL id order. Mixed-width numeric prefixes silently misorder + (`"10"` sorts before `"2"`); the runner WARNS on inconsistent widths — zero-pad ids to equal width (`0001`, + `0002`, …). +- **Stream/type targets are guarded.** A `RenameStream` target in `$`-space or the reserved `mp-migrations` + stream, or a `RenameType` target in `$`-space, is REJECTED up front (such events would be skipped as system + and vanish). Likewise a `DropStream`/`DropStreams` that names a `$`-stream is now HARD-REJECTED — previously + it was a silent no-op (the copy never touches `$`-streams anyway), which could mask a typo. Also note + `RenameType` on a type feeding an `[OutputStream]` projection breaks its linking (below). +- **Schema evolution of history (m10).** `MigrationApplied` is the persisted history record. Adding a + `required` field to it will BREAK deserialization of history written by an older tool version (old records + lack the field). Add new history fields as OPTIONAL (nullable / defaulted), never `required`. + +## `[OutputStream]` ordering — the paced PROJECTION COPY (opt-in) + +The app's `[OutputStream]` merge stream X is fed by a JOIN projection +`fromStreams(['$et-A','$et-B']).linkTo('X')`. Catching that up over a **backlog** that spans multiple `$et` +streams **TYPE-CLUSTERS** the output (drains all `$et-A`, then all `$et-B`): `[A0,B1,A2,B3,A4] → 0,2,4,1,3` +(proven by `Spike1_FromStreams_join_ordering_interleaved_vs_type_clustered`). For an order-SENSITIVE read +model (e.g. the ERP `OffersReadModel` state machine) that corrupts the final state, so the merge stream must +be rebuilt in **commit order**. + +**The fix lives entirely in this tool** (no framework change): the optional `ProjectionCopyContext` turns on +the **paced projection copy** — + +1. Pre-create the app's join projections on the EMPTY dest (discover the source's user projections, read each + query VERBATIM over HTTP `GET /projection/{name}/query`, `CreateContinuous→Disable→Update(emit:true)→Enable`, + then stamp `mp_query_hash = SHA256(query)` on the projection's stream so the app **NO-OPs on boot**). +2. Copy events in `$all` order but **PACED**: after each kept event, wait until the join projection(s) have + emitted its link (the merge stream's link count advances by 1) BEFORE writing the next event. The projection + therefore processes events **one at a time, in commit order, and never accumulates a backlog** — so the + type-clustering (a backlog-catch-up property) can never happen. Final drain to head. + +The pacing keys on the **actual link emission**, not a timer, and is **bounded**: if a projection doesn't emit +a link within the per-event timeout it FAILS LOUD (aborts the run, naming the stalled projection/stream/event) +rather than hanging. + +Determinism is the bar: `Paced_projection_copy_preserves_interleaved_order_deterministically_15x` runs the +interleaved recovery **15×** and requires **15/15** in arrival order; the end-to-end +`Paced_projection_copy_recovers_tombstoned_store_in_commit_order_without_NRE` proves the full `:5081` recovery +(drop the tombstoned stream, rebuild the merge stream in commit order, read model replays with no NRE, app +boots NO-OP). + +> **PERFORMANCE — read before pointing this at a huge store.** Strict per-event pacing is **O(n) SEQUENTIAL**: +> one link-emission round-trip per linkable event. For the ERP (~1070 events) that is seconds-to-minutes — +> fine (correctness over speed). It is **NOT** suitable for very large stores (e.g. millions of events) without +> **batched pacing** (pace per small batch with a per-batch drain — trades a tiny clustering risk for speed). +> That batched mode is a future optimization and is deliberately NOT implemented; strict per-event pacing is +> what gives the determinism guarantee above. + +Without a `ProjectionCopyContext` the tool runs the **simple copy**: it SKIPS the `$>` merge/link streams and +lets the app regenerate them on boot — correct ONLY if the app's projection is itself commit-ordered, which +`fromStreams` is not for a multi-`$et` merge. Use the projection copy whenever merge-stream order matters. + +### Library limitation — `RenameType` vs `[OutputStream]` selectors + +A migration that **renames an event type** feeding an `[OutputStream]` projection breaks that projection's +linking: the projection selects the OLD `$et-{oldType}`, but the migrated events now carry the new type and +land in `$et-{newType}`, so they are never linked into the output stream. Not an issue for the `:5081` +tombstone fix (`_0001` is `DropStreams`-only), but a real limitation to weigh before using `RenameType` on +a type that feeds a merge/read-model stream. + +## Exit codes + +`0` ok · `1` verification mismatch (count OR write-fidelity checksum) · `2` bad args · `3` checksum guard +tripped. (Unparseable-JSON payloads are copied verbatim and do NOT fail the run.) diff --git a/src/MicroPlumberd.Migration.Scripting/JsRuleHost.cs b/src/MicroPlumberd.Migration.Scripting/JsRuleHost.cs new file mode 100644 index 0000000..6b46320 --- /dev/null +++ b/src/MicroPlumberd.Migration.Scripting/JsRuleHost.cs @@ -0,0 +1,441 @@ +using System.Text.Json.Nodes; +using Acornima; +using Jint; +using Jint.Native; +using Jint.Native.Object; +using Jint.Runtime; +using Jint.Runtime.Descriptors; +using Jint.Runtime.Interop; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MicroPlumberd.Migration.Scripting; + +/// +/// Loads a rewrite script into a sandboxed Jint engine and compiles it into +/// operations. +/// +/// +/// The contract is Kurrent Replicator's, so a script written for either tool runs on both: the +/// file may define function transform(original), receiving +/// {Stream, EventType, Data, Metadata, EventId, EventNumber, Created} and returning the same shape, or +/// undefined / an empty Stream / an empty EventType to DROP the event. On top of that the +/// helpers dropStream, dropEvent, update, updateById, renameType, +/// renameStream and log.* keep the common repairs to one line each. +/// Ordering. dropStream, renameStream and renameType are declarative: they +/// are collected while the script is loaded and registered as native builder operations, in that order, +/// BEFORE the single generic Transform that carries everything else. So a dropped stream never reaches +/// update or transform at all, and within one event the order is: dropEvent predicates → +/// update/updateById handlers → transform, which therefore sees the survivors, already +/// updated. +/// Sandbox. CLR interop is left OFF (Jint's default): the script cannot reach a .NET type, +/// the filesystem or the network. Recursion is capped, and every event carries one wall-clock budget +/// (). +/// Threading. A Jint engine is single-threaded and so is this host — which matches the copy +/// engine's single-threaded event loop. Do not share one instance across concurrent runs. +/// +internal sealed class JsRuleHost +{ + /// Wall-clock budget for ALL script code run for one event. + private static readonly TimeSpan PerEventBudget = TimeSpan.FromSeconds(2); + + /// Max nested JS calls — a runaway recursive helper fails instead of taking the process down. + private const int MaxRecursion = 64; + + private readonly Engine _engine; + private readonly PerEventBudgetConstraint _budget = new(PerEventBudget); + private readonly ILogger _logger; + + private readonly List> _streamDrops = []; + private readonly List<(string Old, string New)> _streamRenames = []; + private readonly List<(string Old, string New)> _typeRenames = []; + private readonly List _dropEvents = []; + private readonly List<(string Type, JsValue Fn)> _updatesByType = []; + private readonly List<(string Id, JsValue Fn)> _updatesById = []; + private JsValue? _transform; + + private readonly JsValue _jsonParse; + private readonly JsValue _jsonStringify; + + /// True once the script has been loaded; helper registration is closed after that. + private bool _loaded; + + public JsRuleHost(string source, ILogger? logger = null) + { + ArgumentNullException.ThrowIfNull(source); + _logger = logger ?? NullLogger.Instance; + + _engine = new Engine(o => o + .LimitRecursion(MaxRecursion) + .Constraint(_budget)); + // NOTE: no AllowClr(), no modules, no host functions beyond the helpers below — the script's whole + // world is ECMAScript plus what Register() puts in front of it. + + var json = _engine.GetValue("JSON"); + _jsonParse = json.Get("parse"); + _jsonStringify = json.Get("stringify"); + + RegisterHelpers(); + + try + { + _engine.Execute(source); + } + catch (ParseErrorException ex) + { + throw new ScriptSyntaxException(ex.LineNumber, ex.Column, ex.Message); + } + catch (JavaScriptException ex) + { + // A script that throws at LOAD time (e.g. a bad dropStream argument) is still an authoring error + // the operator must see before anything is started. + var loc = ex.Location; + throw new ScriptSyntaxException(loc.Start.Line, loc.Start.Column + 1, ex.Message); + } + + var t = _engine.GetValue("transform"); + _transform = t.IsObject() && t.IsCallable() ? t : null; + _loaded = true; + } + + /// True when the script declared no rule at all — a pure copy. + public bool IsPureCopy => + _streamDrops.Count == 0 && _streamRenames.Count == 0 && _typeRenames.Count == 0 && + _dropEvents.Count == 0 && _updatesByType.Count == 0 && _updatesById.Count == 0 && _transform is null; + + /// Registers the script's rules on , in the order documented on the class. + public void Register(IMigrationBuilder b) + { + foreach (var drop in _streamDrops) b.DropStream(drop); + foreach (var (o, n) in _streamRenames) b.RenameStream(o, n); + foreach (var (o, n) in _typeRenames) b.RenameType(o, n); + + var needsPerEvent = _dropEvents.Count > 0 || _updatesByType.Count > 0 || _updatesById.Count > 0 + || _transform is not null; + if (needsPerEvent) b.Transform(Apply); + } + + // ---------------------------------------------------------------- helper registration + + private void RegisterHelpers() + { + Define("dropStream", (_, args) => + { + RequireLoading("dropStream"); + if (args.Length == 0) throw Throw("dropStream(name | RegExp | fn) requires one argument."); + _streamDrops.Add(StreamPredicate(args[0])); + return JsValue.Undefined; + }); + + Define("dropEvent", (_, args) => + { + RequireLoading("dropEvent"); + _dropEvents.Add(RequireFunction(args, 0, "dropEvent(fn)")); + return JsValue.Undefined; + }); + + Define("update", (_, args) => + { + RequireLoading("update"); + _updatesByType.Add((RequireString(args, 0, "update(type, fn)"), RequireFunction(args, 1, "update(type, fn)"))); + return JsValue.Undefined; + }); + + Define("updateById", (_, args) => + { + RequireLoading("updateById"); + _updatesById.Add((RequireString(args, 0, "updateById(id, fn)"), RequireFunction(args, 1, "updateById(id, fn)"))); + return JsValue.Undefined; + }); + + Define("renameType", (_, args) => + { + RequireLoading("renameType"); + _typeRenames.Add((RequireString(args, 0, "renameType(from, to)"), RequireString(args, 1, "renameType(from, to)"))); + return JsValue.Undefined; + }); + + Define("renameStream", (_, args) => + { + RequireLoading("renameStream"); + _streamRenames.Add((RequireString(args, 0, "renameStream(from, to)"), RequireString(args, 1, "renameStream(from, to)"))); + return JsValue.Undefined; + }); + + _engine.SetValue("log", ScriptLog.Create(_engine, _logger)); + } + + private void Define(string name, Func fn) => + _engine.SetValue(name, new ClrFunction(_engine, name, fn)); + + /// + /// The declarative helpers describe the RULE SET, which is fixed once the file has been loaded. Calling one + /// from inside transform would silently do nothing (the plan is already compiled), so it fails loudly. + /// + private void RequireLoading(string name) + { + if (_loaded) + throw Throw($"{name}() may only be called at the top level of the script, not while events are " + + "being processed — the rule set is fixed once the script has been loaded."); + } + + private Func StreamPredicate(JsValue arg) + { + if (arg.IsString()) + { + var name = arg.AsString(); + return s => string.Equals(s, name, StringComparison.Ordinal); + } + if (arg.IsRegExp()) + { + var re = arg.AsObject(); + var test = re.Get("test"); + return s => + { + // A /g/ or /y/ regexp carries lastIndex across calls, so the SAME pattern would match every + // other stream. Reset it: a rule must not depend on how many streams preceded this one. + re.Set("lastIndex", 0); + return TypeConverter.ToBoolean(_engine.Invoke(test, re, [JsString.Create(s)])); + }; + } + if (arg.IsCallable()) + { + var fn = arg; + return s => TypeConverter.ToBoolean(_engine.Invoke(fn, JsValue.Undefined, [JsString.Create(s)])); + } + throw Throw("dropStream() accepts a stream name, a RegExp, or a function of the stream name."); + } + + private JsValue RequireFunction(JsValue[] args, int i, string usage) + { + if (args.Length <= i || !args[i].IsCallable()) throw Throw($"{usage}: argument {i + 1} must be a function."); + return args[i]; + } + + private string RequireString(JsValue[] args, int i, string usage) + { + if (args.Length <= i || !args[i].IsString()) throw Throw($"{usage}: argument {i + 1} must be a string."); + var s = args[i].AsString(); + if (s.Length == 0) throw Throw($"{usage}: argument {i + 1} must not be empty."); + return s; + } + + private JavaScriptException Throw(string message) => + new(_engine.Intrinsics.Error, message); + + // ---------------------------------------------------------------- per-event evaluation + + /// + /// The single generic rule the script compiles to. Returns null to drop, the SAME + /// instance when nothing changed (so the copy engine writes the original bytes + /// verbatim), or a rewritten one. + /// + private RawEvent? Apply(RawEvent e) + { + // Only JSON payloads are transformable (requirements). A binary / unparseable payload bypasses the + // script entirely and is copied verbatim — never dropped by a rule that could not even see it. + if (e.Data is null) return e; + + _budget.BeginEvent(); + try + { + var current = ToJsEvent(e); + // Snapshot the payload AS JAVASCRIPT SEES IT, before any rule runs. Comparing the script's result + // against this — rather than against the .NET node's own text — is what makes "the script changed + // nothing" decidable: the trip through JS canonicalises numbers (1.0 renders as 1), so a .NET-side + // comparison reports every such payload as changed and re-serialises a store that nobody edited. + var dataIn = Snapshot(current.Get("Data")); + var metaIn = Snapshot(current.Get("Metadata")); + + foreach (var pred in _dropEvents) + if (TypeConverter.ToBoolean(_engine.Invoke(pred, JsValue.Undefined, [current]))) + return null; + + foreach (var (type, fn) in _updatesByType) + { + if (!string.Equals(current.Get("EventType").ToString(), type, StringComparison.Ordinal)) continue; + var next = _engine.Invoke(fn, JsValue.Undefined, [current]); + if (IsDrop(next)) return null; + current = RequireEventObject(next, e, "update"); + } + + foreach (var (id, fn) in _updatesById) + { + if (!string.Equals(e.EventId.ToString(), id, StringComparison.OrdinalIgnoreCase)) continue; + var next = _engine.Invoke(fn, JsValue.Undefined, [current]); + if (IsDrop(next)) return null; + current = RequireEventObject(next, e, "updateById"); + } + + if (_transform is not null) + { + var next = _engine.Invoke(_transform, JsValue.Undefined, [current]); + if (IsDrop(next)) return null; + current = RequireEventObject(next, e, "transform"); + } + + return FromJsEvent(current, e, dataIn, metaIn); + } + catch (ScriptExecutionException) + { + throw; + } + catch (TimeoutException ex) + { + throw new ScriptExecutionException(e.StreamId, e.EventNumber, + $"{ex.Message} An event-level budget is the only defence against a script that never returns.", ex); + } + catch (JavaScriptException ex) + { + throw new ScriptExecutionException(e.StreamId, e.EventNumber, ex.Message, ex); + } + catch (JintException ex) + { + throw new ScriptExecutionException(e.StreamId, e.EventNumber, ex.Message, ex); + } + finally + { + _budget.EndEvent(); + } + } + + private static bool IsDrop(JsValue v) => v.IsUndefined() || v.IsNull(); + + private static ObjectInstance RequireEventObject(JsValue v, RawEvent e, string what) + { + if (!v.IsCallable() && v is ObjectInstance o) return o; + throw new ScriptExecutionException(e.StreamId, e.EventNumber, + $"{what}() must return an event object (or undefined to drop) — it returned {Describe(v)}."); + } + + private static string Describe(JsValue v) => + v.IsString() ? $"the string \"{v.AsString()}\"" : + v.IsNumber() ? $"the number {v.AsNumber()}" : + v.IsBoolean() ? $"the boolean {v.AsBoolean()}" : + v.IsCallable() ? "a function" : + v.Type.ToString().ToLowerInvariant(); + + // ---------------------------------------------------------------- marshalling + + /// + /// Builds the script's view of one event. Data/Metadata cross as real JS objects through the + /// engine's own JSON.parse — never JsValue.FromObject on a , which would + /// hand the script a CLR wrapper whose members are .NET's, not JavaScript's. + /// + private ObjectInstance ToJsEvent(RawEvent e) + { + var o = new JsObject(_engine); + o.FastSetDataProperty("Stream", e.StreamId); + o.FastSetDataProperty("EventType", e.Type); + o.FastSetDataProperty("Data", ToJs(e.Data)); + o.FastSetDataProperty("Metadata", ToJs(e.Metadata)); + // Identity, read-only: the copy preserves the source id, renumbers the destination stream, and stamps + // its own write time — assigning to these could only mislead the script's author. + o.FastSetProperty("EventId", ReadOnly(JsString.Create(e.EventId.ToString()))); + o.FastSetProperty("EventNumber", ReadOnly(JsNumber.Create((double)e.EventNumber))); + // ISO-8601 so the documented `e.Created < "2026-08-25"` string comparison orders correctly. + o.FastSetProperty("Created", ReadOnly(JsString.Create( + e.Created.ToUniversalTime().ToString("O", System.Globalization.CultureInfo.InvariantCulture)))); + return o; + } + + /// + /// Reads Stream / EventType off the returned object, insisting it is a string. + /// + /// + /// Absent is NOT empty. Building a fresh result object and forgetting a field is the most likely mistake a + /// script author makes, and the contract's "an empty Stream drops the event" would otherwise turn it + /// into total, silent data loss at exit 0. A non-string is rejected for the same reason from the other + /// side: coercing 123 to "123" writes events into a stream nobody named. Neither is a drop, + /// so both are errors — and a script that means "drop" still writes undefined or "", which is + /// what Replicator documents. + /// + private static string RequiredString(ObjectInstance o, string property, RawEvent e) + { + var v = o.Get(property); + if (v.IsUndefined() || v.IsNull()) + throw new ScriptExecutionException(e.StreamId, e.EventNumber, + $"the returned event has no '{property}'. That is not the same as an empty '{property}', which " + + "would DROP the event — return the original's value, or an explicit empty string if dropping " + + "is what you meant."); + if (!v.IsString()) + throw new ScriptExecutionException(e.StreamId, e.EventNumber, + $"the returned event's '{property}' is {Describe(v)}, but it must be a string."); + return v.AsString(); + } + + /// The JSON text of a value as JavaScript renders it, or null for undefined/null. + private string? Snapshot(JsValue v) + { + if (v.IsUndefined() || v.IsNull()) return null; + var text = _engine.Invoke(_jsonStringify, JsValue.Undefined, [v]); + return text.IsString() ? text.AsString() : null; + } + + private static PropertyDescriptor ReadOnly(JsValue v) => + new(v, writable: false, enumerable: true, configurable: false); + + private JsValue ToJs(JsonNode? node) => + node is null ? JsValue.Undefined : _engine.Invoke(_jsonParse, JsValue.Undefined, [JsString.Create(node.ToJsonString())]); + + /// + /// Reads the script's result back. Returns the ORIGINAL instance when nothing + /// changed, so the copy engine takes its byte-verbatim path. + /// + private RawEvent? FromJsEvent(ObjectInstance o, RawEvent original, string? dataIn, string? metaIn) + { + var streamName = RequiredString(o, "Stream", original); + var typeName = RequiredString(o, "EventType", original); + + // The Replicator contract's second way of saying "drop" — an EXPLICIT empty string. An absent or + // non-string property is an authoring error and was rejected above; folding it in here is what would + // turn one forgotten field into the silent deletion of every event the rule touched. + if (streamName.Length == 0 || typeName.Length == 0) return null; + + var data = FromJs(o.Get("Data"), original.Data, dataIn, original, "Data"); + var meta = FromJs(o.Get("Metadata"), original.Metadata, metaIn, original, "Metadata"); + + if (ReferenceEquals(data, original.Data) && ReferenceEquals(meta, original.Metadata) + && string.Equals(streamName, original.StreamId, StringComparison.Ordinal) + && string.Equals(typeName, original.Type, StringComparison.Ordinal)) + return original; + + return original with { StreamId = streamName, Type = typeName, Data = data, Metadata = meta }; + } + + /// + /// Converts one JS value back to a via the engine's JSON.stringify, returning + /// UNCHANGED (same reference) when the round-trip is textually identical — which + /// is what lets an untouched payload be copied byte-for-byte instead of re-rendered. + /// + private JsonNode? FromJs(JsValue v, JsonNode? original, string? snapshot, RawEvent e, string what) + { + if (v.IsUndefined() || v.IsNull()) + { + // A script that CLEARS metadata means it: return an empty object rather than null, which the copy + // engine would read as "unchanged" and silently write the original metadata back. + if (original is null) return null; + return what == "Metadata" ? new JsonObject() : throw new ScriptExecutionException( + e.StreamId, e.EventNumber, "Data must not be removed — return undefined to drop the event."); + } + + var text = _engine.Invoke(_jsonStringify, JsValue.Undefined, [v]); + if (!text.IsString()) + throw new ScriptExecutionException(e.StreamId, e.EventNumber, + $"{what} is not JSON-serialisable (JSON.stringify returned {Describe(text)})."); + + var json = text.AsString(); + if (original is not null && snapshot is not null && string.Equals(json, snapshot, StringComparison.Ordinal)) + return original; + + try + { + return JsonNode.Parse(json); + } + catch (System.Text.Json.JsonException ex) + { + throw new ScriptExecutionException(e.StreamId, e.EventNumber, + $"{what} could not be read back as JSON: {ex.Message}", ex); + } + } +} diff --git a/src/MicroPlumberd.Migration.Scripting/MicroPlumberd.Migration.Scripting.csproj b/src/MicroPlumberd.Migration.Scripting/MicroPlumberd.Migration.Scripting.csproj new file mode 100644 index 0000000..67997df --- /dev/null +++ b/src/MicroPlumberd.Migration.Scripting/MicroPlumberd.Migration.Scripting.csproj @@ -0,0 +1,39 @@ + + + + net10.0 + enable + enable + embedded + true + JavaScript rule scripting for MicroPlumberd.Migration: loads a Kurrent-Replicator-compatible rewrite script (function transform(original) plus dropStream/dropEvent/update/updateById/renameType/renameStream helpers) into a sandboxed Jint engine and compiles it to a Migration. + MicroPlumberd.Migration.Scripting + Rafal Maciag + Rafal Maciag + logo-squere.png + README.md + MIT + https://github.com/modelingevolution/micro-plumberd + https://modelingevolution.github.io/micro-plumberd/ + EventStore;KurrentDB;CQRS;EventSourcing;Migration;JavaScript;Jint + + + + + + + + + + + + + + + + + + + + diff --git a/src/MicroPlumberd.Migration.Scripting/PerEventBudgetConstraint.cs b/src/MicroPlumberd.Migration.Scripting/PerEventBudgetConstraint.cs new file mode 100644 index 0000000..b2273b4 --- /dev/null +++ b/src/MicroPlumberd.Migration.Scripting/PerEventBudgetConstraint.cs @@ -0,0 +1,44 @@ +using System.Diagnostics; +using Jint; + +namespace MicroPlumberd.Migration.Scripting; + +/// +/// Caps the total JavaScript execution time spent on ONE event, across every helper and the +/// transform call together. +/// +/// +/// Jint's own TimeoutInterval is per Execute/Invoke entry and is RESET at each +/// one — with a dropEvent, two updates and a transform that is four separate budgets, +/// so a script could spend four times the limit on every event and never trip. This constraint is armed once +/// per event by and deliberately does NOT reset on re-entry, which is the whole +/// point: the budget belongs to the event, not to the call. +/// is called by the interpreter on statement boundaries, so a tight +/// while(true){} is interrupted; it cannot interrupt a single non-yielding host operation. +/// +internal sealed class PerEventBudgetConstraint(TimeSpan budget) : Constraint +{ + private long _deadline = long.MaxValue; + + /// Arms a fresh budget for the event about to be processed. + public void BeginEvent() => + _deadline = Stopwatch.GetTimestamp() + (long)(budget.TotalSeconds * Stopwatch.Frequency); + + /// Disarms the budget (no script is running). + public void EndEvent() => _deadline = long.MaxValue; + + public override void Check() + { + if (Stopwatch.GetTimestamp() > _deadline) + throw new TimeoutException( + $"Script exceeded its per-event time budget of {budget.TotalSeconds:0.###}s."); + } + + /// + /// Intentionally a NO-OP. Jint resets constraints on every script entry; resetting here would restart the + /// budget for each helper call and defeat the per-event cap. + /// + public override void Reset() + { + } +} diff --git a/src/MicroPlumberd.Migration.Scripting/ScriptExceptions.cs b/src/MicroPlumberd.Migration.Scripting/ScriptExceptions.cs new file mode 100644 index 0000000..2d400a0 --- /dev/null +++ b/src/MicroPlumberd.Migration.Scripting/ScriptExceptions.cs @@ -0,0 +1,33 @@ +namespace MicroPlumberd.Migration.Scripting; + +/// +/// The script could not be PARSED. Thrown by the constructor, i.e. before any +/// store is touched, so a typo can never leave a half-rewritten system behind. +/// +public sealed class ScriptSyntaxException(int line, int column, string description) + : Exception($"Script syntax error at line {line}, column {column}: {description}") +{ + /// 1-based line of the offending token. + public int Line { get; } = line; + + /// 1-based column of the offending token. + public int Column { get; } = column; + + /// The parser's own description of the problem. + public string Description { get; } = description; +} + +/// +/// A script rule THREW, ran past its per-event time budget, or returned something the contract does not +/// allow, while processing one specific event — which the message names, because "the script failed" is +/// useless to an operator holding a store with a million events. +/// +public sealed class ScriptExecutionException(string stream, ulong eventNumber, string reason, Exception? inner = null) + : Exception($"Script failed on {stream}#{eventNumber}: {reason}", inner) +{ + /// The stream of the event being processed. + public string Stream { get; } = stream; + + /// The source event number of the event being processed. + public ulong EventNumber { get; } = eventNumber; +} diff --git a/src/MicroPlumberd.Migration.Scripting/ScriptLog.cs b/src/MicroPlumberd.Migration.Scripting/ScriptLog.cs new file mode 100644 index 0000000..f79c4cd --- /dev/null +++ b/src/MicroPlumberd.Migration.Scripting/ScriptLog.cs @@ -0,0 +1,53 @@ +using Jint; +using Jint.Native; +using Jint.Native.Object; +using Jint.Runtime.Interop; +using Microsoft.Extensions.Logging; + +namespace MicroPlumberd.Migration.Scripting; + +/// +/// Builds the script-visible log object (log.debug/info/warn/error), bridging a +/// Serilog-style message template plus values onto . +/// +/// +/// The object is assembled from Jint s rather than by handing Jint a CLR instance: +/// wrapping a CLR object would expose its whole reflection surface (GetType(), and from there the +/// loaded assemblies) to the script, which is exactly what the sandbox exists to prevent. +/// +internal static class ScriptLog +{ + public static ObjectInstance Create(Engine engine, ILogger logger) + { + var log = new JsObject(engine); + log.FastSetDataProperty("debug", Level(engine, logger, LogLevel.Debug, "debug")); + log.FastSetDataProperty("info", Level(engine, logger, LogLevel.Information, "info")); + log.FastSetDataProperty("warn", Level(engine, logger, LogLevel.Warning, "warn")); + log.FastSetDataProperty("error", Level(engine, logger, LogLevel.Error, "error")); + return log; + } + + private static ClrFunction Level(Engine engine, ILogger logger, LogLevel level, string name) => + new(engine, name, (_, args) => + { + if (args.Length == 0) return JsValue.Undefined; + var template = args[0].IsString() ? args[0].AsString() : args[0].ToString(); + var values = new object?[args.Length - 1]; + for (var i = 1; i < args.Length; i++) values[i - 1] = ToClr(args[i]); +#pragma warning disable CA2254 // the template IS the script author's — that is the feature + logger.Log(level, template, values); +#pragma warning restore CA2254 + return JsValue.Undefined; + }); + + // Only PRIMITIVES cross as themselves; anything structural is rendered as JSON-ish text via the engine's + // own ToString. ToObject() on an object would hand the logger a Jint wrapper whose ToString is useless. + private static object? ToClr(JsValue v) => v switch + { + _ when v.IsNull() || v.IsUndefined() => null, + _ when v.IsString() => v.AsString(), + _ when v.IsBoolean() => v.AsBoolean(), + _ when v.IsNumber() => v.AsNumber(), + _ => v.ToString() + }; +} diff --git a/src/MicroPlumberd.Migration.Scripting/ScriptMigration.cs b/src/MicroPlumberd.Migration.Scripting/ScriptMigration.cs new file mode 100644 index 0000000..48d8b12 --- /dev/null +++ b/src/MicroPlumberd.Migration.Scripting/ScriptMigration.cs @@ -0,0 +1,65 @@ +using System.Security.Cryptography; +using System.Text; +using Microsoft.Extensions.Logging; + +namespace MicroPlumberd.Migration.Scripting; + +/// +/// A whose rules come from a JavaScript file rather than from C# code. +/// +/// +/// The script is PARSED IN THE CONSTRUCTOR. That is deliberate: a rewrite tool must reject a typo +/// before it starts a container, stops a store or renames a directory, so construction is the last cheap +/// moment to fail ( carries the line and column). +/// is {idPrefix}_{sha256(source)[..12]} — the prefix says WHEN/WHY the rewrite +/// ran, the hash says WHICH script did it, and the two together make each run its own history record. +/// +public sealed class ScriptMigration : Migration +{ + private readonly JsRuleHost _host; + + /// Loads and parses . + /// The script text (the Replicator contract plus this library's helpers). + /// + /// Stable, sortable prefix for — e.g. rewrite_20260907T151200. Two runs of the + /// SAME script must not share an id unless they are meant to be the same migration: the history guard + /// skips an id it has already applied. + /// + /// Sink for the script's log.* calls. + /// The script does not parse. + public ScriptMigration(string source, string idPrefix, ILogger? logger = null) + { + ArgumentNullException.ThrowIfNull(source); + ArgumentException.ThrowIfNullOrWhiteSpace(idPrefix); + + Source = source; + ScriptChecksum = Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(source))); + Id = $"{idPrefix}_{ScriptChecksum[..12]}"; + _host = new JsRuleHost(source, logger); + } + + /// The script text this migration was built from. + public string Source { get; } + + /// Uppercase hex SHA-256 of the script text — the identity of these rules. + public string ScriptChecksum { get; } + + /// + public override string Id { get; } + + /// + /// The script's own hash, NOT the operation-descriptor checksum: every script compiles to the same host + /// delegates, so descriptors cannot tell two scripts apart. See . + /// + public override string? ChecksumOverride => ScriptChecksum; + + /// True when the script declares no rule at all — the run is a pure copy. + public bool IsPureCopy => _host.IsPureCopy; + + /// + public override void Migrate(IMigrationBuilder b) + { + ArgumentNullException.ThrowIfNull(b); + _host.Register(b); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/ChecksumAndDiscoveryTests.cs b/src/MicroPlumberd.Migration.Tests/ChecksumAndDiscoveryTests.cs new file mode 100644 index 0000000..e6b988d --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/ChecksumAndDiscoveryTests.cs @@ -0,0 +1,73 @@ +using System.Reflection; +using FluentAssertions; +using MicroPlumberd.Migration; +using Xunit; + +namespace MicroPlumberd.Migration.Tests; + +public class ChecksumAndDiscoveryTests +{ + [Fact] + public void Checksum_is_stable_across_recompiles_of_the_same_rules() + { + var a = CompiledMigration.Compile(new TestMigrations.DropManyStreams()); + var b = CompiledMigration.Compile(new TestMigrations.DropManyStreams()); + a.Checksum.Should().Be(b.Checksum); + } + + [Fact] + public void Checksum_changes_when_literal_arguments_change() + { + var a = CompiledMigration.Compile(new InlineMigration("x", b => b.DropStream("Foo-1"))); + var b = CompiledMigration.Compile(new InlineMigration("x", b => b.DropStream("Foo-2"))); + a.Checksum.Should().NotBe(b.Checksum); + } + + [Fact] + public void Checksum_changes_when_the_set_of_operations_changes() + { + var a = CompiledMigration.Compile(new InlineMigration("x", b => b.DropStream("Foo-1"))); + var b = CompiledMigration.Compile(new InlineMigration("x", b => + { + b.DropStream("Foo-1"); + b.RenameType("A", "B"); + })); + a.Checksum.Should().NotBe(b.Checksum); + } + + [Fact] + public void Checksum_changes_when_a_lambda_body_changes() + { + // Two predicates with different bodies must hash differently (IL-sensitive checksum). + var a = CompiledMigration.Compile(new InlineMigration("x", b => b.DropEvent(e => e.Type == "One"))); + var b = CompiledMigration.Compile(new InlineMigration("x", b => b.DropEvent(e => e.Type == "Two"))); + a.Checksum.Should().NotBe(b.Checksum); + } + + [Fact] + public void Discovery_orders_migrations_by_id_and_includes_the_shipped_one() + { + var runnerAssembly = typeof(Runner.Migrations._0001_DropTombstonedTestOffers).Assembly; + var found = MigrationDiscovery.FromAssemblies(runnerAssembly); + found.Should().ContainSingle(m => m.Id == "0001_drop_tombstoned_test_offers"); + found.Select(m => m.Id).Should().BeInAscendingOrder(StringComparer.Ordinal); + } + + [Fact] + public void Discovery_rejects_duplicate_ids() + { + var migrations = new Migration[] + { + new InlineMigration("dup", _ => { }), + new InlineMigration("dup", _ => { }) + }; + var act = () => MigrationDiscovery.Order(migrations); + act.Should().Throw().Which.MigrationId.Should().Be("dup"); + } + + private sealed class InlineMigration(string id, Action build) : Migration + { + public override string Id => id; + public override void Migrate(IMigrationBuilder b) => build(b); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/FieldPartitionAndDirectSubscribeSpikes.cs b/src/MicroPlumberd.Migration.Tests/FieldPartitionAndDirectSubscribeSpikes.cs new file mode 100644 index 0000000..0867ba7 --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/FieldPartitionAndDirectSubscribeSpikes.cs @@ -0,0 +1,413 @@ +using System.Collections.Concurrent; +using System.Net; +using System.Text; +using System.Text.Json.Nodes; +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Testing; +using Xunit; +using Xunit.Abstractions; + +namespace MicroPlumberd.Migration.Tests; + +/// +/// SPIKE — second empirical batch against REAL KurrentDB 26.1, closing the last design questions: +/// SPIKE-8 FIELD-VALUE PARTITION (the potential shape-B rescue): an index defined with a "fields" selector +/// (rec => rec.value.<field>) exposes a per-value partition stream "$idx-user-<name>:<value>". Does +/// reading/subscribing a single partition deliver ONLY that key's events, in commit order? If yes, +/// per-key lookup (Inbox/Outbox routing) MAY be index-backable and shape-B is NOT permanently excluded. +/// SPIKE-9 DIRECT-SUBSCRIBE DEFINITIVE: exhaustively confirm the whole-index stream "$idx-user-<name>" is NOT +/// directly consumable — SubscribeToStream AND ReadStream, resolveLinkTos both true and false — i.e. the +/// documented recipe is the filtered-$all prefix, nothing else. +/// SPIKE-10 CaughtUp + no-count-gate: on the working filtered SubscribeToAll(prefix) path, does StreamMessage.CaughtUp +/// fire at the history→live boundary (read models need it to mark ready), and does the live path tail +/// WITHOUT any count-based readiness gate (the offline WaitUntilReadyAsync count-gate is offline-only)? +/// +/// NOTE on SPIKE-4 (first batch): that test probed body access via rec.data/rec.body and saw zero — but the +/// KurrentDB 26.1 body accessor is rec.value (per docs). SPIKE-8 here re-tests with the correct accessor + the +/// dedicated "fields" mechanism, and supersedes SPIKE-4's "body not accessible" conclusion where they differ. +/// +[Trait("Category", "Integration")] +public class FieldPartitionAndDirectSubscribeSpikes(ITestOutputHelper output) +{ + private sealed class Store : IAsyncDisposable + { + private readonly List _servers = new(); + public async Task<(KurrentDBClient Client, string Conn)> NewAsync(string tag) + { + var es = EventStoreServer.Create($"mp-fp-{tag}-{Guid.NewGuid():N}"); + _servers.Add(es); + await es.StartInDocker(inMemory: true); + var s = es.GetEventStoreSettings(); + return (new KurrentDBClient(s), es.HttpUrl.ToString()); + } + public async ValueTask DisposeAsync() { foreach (var s in _servers) await s.DisposeAsync(); } + } + + private static EventData Bodied(string type, int ord, string someId) => + new(Uuid.NewUuid(), type, Encoding.UTF8.GetBytes($"{{\"Ord\":{ord},\"SomeId\":\"{someId}\"}}")); + + // Raw filtered-$all read over an arbitrary stream PREFIX (resolveLinkTos), collecting the resolved body "Ord". + private static async Task> ReadPrefixOrdsAsync(KurrentDBClient client, string prefix, CancellationToken ct = default) + { + var ords = new List(); + var read = client.ReadAllAsync(Direction.Forwards, Position.Start, StreamFilter.Prefix(prefix), + maxCount: long.MaxValue, resolveLinkTos: true, cancellationToken: ct); + var e = read.GetAsyncEnumerator(ct); + try + { + while (true) + { + try { if (!await e.MoveNextAsync()) break; } + catch (Grpc.Core.RpcException ex) when (ex.StatusCode == Grpc.Core.StatusCode.NotFound) { break; } + if (e.Current.Event is null) continue; + ords.Add(JsonNode.Parse(e.Current.Event.Data.Span)?["Ord"]?.GetValue() ?? -1); + } + } + finally { await e.DisposeAsync(); } + return ords; + } + + private static async Task WaitUntil(Func> cond, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) { if (await cond()) return true; await Task.Delay(200); } + return await cond(); + } + + // ===================================================================================================== + // SPIKE-8 — FIELD-VALUE PARTITION. Define an index keyed by a body field (selector rec => rec.value.SomeId) + // and confirm the per-value partition stream "$idx-user-:" delivers ONLY that key's events in + // commit order — the shape-B (per-key lookup) rescue test. + // ===================================================================================================== + [Fact] + public async Task Spike8_field_value_partition_delivers_single_key_in_commit_order() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s8"); + + // Interleaved two partitions: alpha gets Ord 0,2,4 ; beta gets 1,3 (distinct values, no prefix overlap). + await client.AppendToStreamAsync("S8-1", StreamState.NoStream, new[] + { + Bodied("UdiC", 0, "alpha"), Bodied("UdiC", 1, "beta"), Bodied("UdiC", 2, "alpha"), + Bodied("UdiC", 3, "beta"), Bodied("UdiC", 4, "alpha") + }); + + var (httpBase, user, pass) = KurrentHttpEndpoint.Parse(conn); + using var http = KurrentHttpEndpoint.CreateClient(user, pass); + + const string name = "orders-by-key"; + var url = new Uri(httpBase, $"v2/indexes/{name}"); + var payload = new JsonObject + { + ["filter"] = "rec => rec.schema.name == 'UdiC'", + ["fields"] = new JsonArray(new JsonObject + { + ["name"] = "someid", + ["selector"] = "rec => rec.value.SomeId", + ["type"] = "INDEX_FIELD_TYPE_STRING" + }), + ["start"] = true + }; + using var content = new StringContent(payload.ToJsonString(), Encoding.UTF8, "application/json"); + var resp = await http.PostAsync(url, content); + var body = await resp.Content.ReadAsStringAsync(); + output.WriteLine($"SPIKE-8 POST (fields=someid rec.value.SomeId) -> HTTP {(int)resp.StatusCode}: {body}"); + resp.IsSuccessStatusCode.Should().BeTrue("a field-keyed index definition must be accepted"); + + var whole = UserDefinedIndexSource.IndexStream(name); // "$idx-user-orders-by-key" + var partAlpha = $"{whole}:alpha"; // "$idx-user-orders-by-key:alpha" + var partBeta = $"{whole}:beta"; + + // Whole index must backfill all 5 (in commit order) — also our readiness signal. + var wholeReady = await WaitUntil(async () => (await ReadPrefixOrdsAsync(client, whole)).Count >= 5, TimeSpan.FromSeconds(30)); + var wholeOrds = await ReadPrefixOrdsAsync(client, whole); + output.WriteLine($"SPIKE-8 whole index (ready={wholeReady}): {string.Join(",", wholeOrds)}"); + + // THE shape-B test: the alpha partition must contain ONLY alpha's events (0,2,4), in commit order. + var alpha = await ReadPrefixOrdsAsync(client, partAlpha); + var beta = await ReadPrefixOrdsAsync(client, partBeta); + output.WriteLine($"SPIKE-8 partition :alpha = {string.Join(",", alpha)} (expect 0,2,4)"); + output.WriteLine($"SPIKE-8 partition :beta = {string.Join(",", beta)} (expect 1,3)"); + + // Live: append more to alpha AFTER building; the partition read must pick it up in commit order. + await client.AppendToStreamAsync("S8-2", StreamState.NoStream, [Bodied("UdiC", 5, "beta"), Bodied("UdiC", 6, "alpha")]); + var alphaLive = await WaitUntil(async () => (await ReadPrefixOrdsAsync(client, partAlpha)).Contains(6), TimeSpan.FromSeconds(30)); + var alpha2 = await ReadPrefixOrdsAsync(client, partAlpha); + output.WriteLine($"SPIKE-8 partition :alpha after live (picked6={alphaLive}) = {string.Join(",", alpha2)} (expect 0,2,4,6)"); + + // ASSERTIONS — encode the shape-B verdict. + wholeOrds.Should().Equal(new[] { 0, 1, 2, 3, 4 }, "whole field-index still reads all events in commit order"); + alpha.Should().Equal(new[] { 0, 2, 4 }, "the :alpha partition delivers ONLY alpha's events, in commit order"); + beta.Should().Equal(new[] { 1, 3 }, "the :beta partition delivers ONLY beta's events, in commit order"); + alpha2.Should().Equal(new[] { 0, 2, 4, 6 }, "the partition live-updates in commit order (shape-B is index-backable)"); + } + + // ===================================================================================================== + // SPIKE-11 — SUBSCRIBE (catch-up→live PUSH) to a SINGLE field partition. SPIKE-8 proved partition CONTENT + // via read; this proves the go/no-go path the design actually uses: an OPEN SubscribeToAll over + // StreamFilter.Prefix("$idx-user-:") delivers ONLY that key's events on catch-up AND pushes only + // that key's live appends (ignoring other keys), in commit order — i.e. shape-B (per-key lookup) is a real + // live subscription, not just a re-readable snapshot. + // ===================================================================================================== + [Fact] + public async Task Spike11_subscribe_single_field_partition_pushes_only_that_key_catchup_and_live() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s11"); + + await client.AppendToStreamAsync("S11-1", StreamState.NoStream, new[] + { + Bodied("UdiC", 0, "alpha"), Bodied("UdiC", 1, "beta"), Bodied("UdiC", 2, "alpha"), + Bodied("UdiC", 3, "beta"), Bodied("UdiC", 4, "alpha") + }); + + var (httpBase, user, pass) = KurrentHttpEndpoint.Parse(conn); + using var http = KurrentHttpEndpoint.CreateClient(user, pass); + const string name = "s11-by-key"; + var url = new Uri(httpBase, $"v2/indexes/{name}"); + var payload = new JsonObject + { + ["filter"] = "rec => rec.schema.name == 'UdiC'", + ["fields"] = new JsonArray(new JsonObject + { ["name"] = "someid", ["selector"] = "rec => rec.value.SomeId", ["type"] = "INDEX_FIELD_TYPE_STRING" }), + ["start"] = true + }; + using var content = new StringContent(payload.ToJsonString(), Encoding.UTF8, "application/json"); + (await http.PostAsync(url, content)).EnsureSuccessStatusCode(); + + var whole = UserDefinedIndexSource.IndexStream(name); + var partAlpha = $"{whole}:alpha"; + // wait until the alpha partition has backfilled its 3 (via read) before opening the subscription + await WaitUntil(async () => (await ReadPrefixOrdsAsync(client, partAlpha)).Count >= 3, TimeSpan.FromSeconds(30)); + + var received = new ConcurrentQueue(); + var caughtUp = false; + using var cts = new CancellationTokenSource(); + var pump = Task.Run(async () => + { + var filter = new SubscriptionFilterOptions(StreamFilter.Prefix(partAlpha)); + await using var sub = client.SubscribeToAll(FromAll.Start, resolveLinkTos: true, filterOptions: filter, + cancellationToken: cts.Token); + try + { + await foreach (var m in sub.Messages.WithCancellation(cts.Token)) + switch (m) + { + case StreamMessage.Event(var e): + received.Enqueue(JsonNode.Parse(e.Event.Data.Span)?["Ord"]?.GetValue() ?? -1); + break; + case StreamMessage.CaughtUp: caughtUp = true; break; + } + } + catch (OperationCanceledException) { } + }); + + var gotHistory = await WaitUntil(() => Task.FromResult(received.Count >= 3), TimeSpan.FromSeconds(30)); + output.WriteLine($"SPIKE-11 partition :alpha catch-up (n={received.Count} caughtUp={caughtUp}): {string.Join(",", received)} (expect 0,2,4)"); + + // Live: append to BOTH keys. The alpha subscription must push ONLY 6 (alpha), NOT 5 (beta). + await client.AppendToStreamAsync("S11-2", StreamState.NoStream, [Bodied("UdiC", 5, "beta"), Bodied("UdiC", 6, "alpha")]); + var gotLive = await WaitUntil(() => Task.FromResult(received.Contains(6)), TimeSpan.FromSeconds(30)); + await Task.Delay(1000); // give any erroneous beta(5) delivery a chance to show up + + cts.Cancel(); await pump; + + output.WriteLine($"SPIKE-11 partition :alpha after live (gotLive={gotLive}): {string.Join(",", received)} (expect 0,2,4,6 — NO 5)"); + gotHistory.Should().BeTrue("the partition subscription must catch up over that key's history"); + gotLive.Should().BeTrue("the partition subscription must PUSH that key's live appends"); + received.Should().Equal(new[] { 0, 2, 4, 6 }, + "the partition subscription delivers ONLY alpha (0,2,4,6) — never beta's 1,3,5 — in commit order"); + received.Should().NotContain(5, "beta's live append must NOT leak into the alpha partition subscription"); + } + + // ===================================================================================================== + // SPIKE-12 ($ce-category / stream-selection filter expressivity) — a shape-A projection built on + // fromCategory('X') / $ce-X selects by STREAM CATEGORY, not event type. Can a user-defined index filter + // express that? Docs (features/indexes/user-defined.html) expose the stream name as rec.position.stream (no + // dedicated category property; category = prefix before '-'). Test category-style filters on rec.position.stream + // against a schema-name CONTROL that MUST match (so an empty category result is a real FAIL, not a broken harness). + // If PASS: EnsureFieldPartitionedIndexAsync can use a direct category filter. If FAIL: it must enumerate the + // category's known event types (with the "misses a type added later" caveat). + // ===================================================================================================== + [Fact] + public async Task Spike12_category_stream_selection_filter_expressivity() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s12"); + + // SAME event type "Ev" in TWO categories so ONLY the stream/category can discriminate them. + // Category X (streams "X-..") gets Ord 0,2 ; category Y gets Ord 1,3 — interleaved in $all commit order. + await client.AppendToStreamAsync("X-1", StreamState.NoStream, + [new EventData(Uuid.NewUuid(), "Ev", Encoding.UTF8.GetBytes("{\"Ord\":0}"))]); + await client.AppendToStreamAsync("Y-1", StreamState.NoStream, + [new EventData(Uuid.NewUuid(), "Ev", Encoding.UTF8.GetBytes("{\"Ord\":1}"))]); + await client.AppendToStreamAsync("X-2", StreamState.NoStream, + [new EventData(Uuid.NewUuid(), "Ev", Encoding.UTF8.GetBytes("{\"Ord\":2}"))]); + await client.AppendToStreamAsync("Y-2", StreamState.NoStream, + [new EventData(Uuid.NewUuid(), "Ev", Encoding.UTF8.GetBytes("{\"Ord\":3}"))]); + + var (httpBase, user, pass) = KurrentHttpEndpoint.Parse(conn); + using var http = KurrentHttpEndpoint.CreateClient(user, pass); + + async Task<(HttpStatusCode Status, string Body, List Ords)> TryFilter(string name, string filter, int expectMatches) + { + var url = new Uri(httpBase, $"v2/indexes/{name}"); + using var content = new StringContent(new JsonObject { ["filter"] = filter, ["start"] = true }.ToJsonString(), + Encoding.UTF8, "application/json"); + var r = await http.PostAsync(url, content); + var body = await r.Content.ReadAsStringAsync(); + var ords = new List(); + if (r.IsSuccessStatusCode) + { + await WaitUntil(async () => (await ReadPrefixOrdsAsync(client, UserDefinedIndexSource.IndexStream(name))).Count >= expectMatches, TimeSpan.FromSeconds(15)); + ords = await ReadPrefixOrdsAsync(client, UserDefinedIndexSource.IndexStream(name)); + } + return (r.StatusCode, body, ords); + } + + // CONTROL — schema-name filter that MUST match all 4 (proves the harness/index works on this store). + var ctrl = await TryFilter("s12-ctrl", "rec => rec.schema.name == 'Ev'", 4); + output.WriteLine($"SPIKE-12 CONTROL rec.schema.name=='Ev' -> HTTP {(int)ctrl.Status}: ords={string.Join(",", ctrl.Ords)} (expect 0,1,2,3)"); + + // CATEGORY filters on rec.position.stream — two forms (startsWith, and split-before-dash). + var startsWith = await TryFilter("s12-startswith", "rec => rec.position.stream.startsWith('X-')", 2); + output.WriteLine($"SPIKE-12 rec.position.stream.startsWith('X-') -> HTTP {(int)startsWith.Status}: {(startsWith.Status == HttpStatusCode.OK ? $"ords={string.Join(",", startsWith.Ords)}" : startsWith.Body)} (expect 0,2 if category-selection works)"); + + var split = await TryFilter("s12-split", "rec => rec.position.stream.split('-')[0] == 'X'", 2); + output.WriteLine($"SPIKE-12 rec.position.stream.split('-')[0]=='X' -> HTTP {(int)split.Status}: {(split.Status == HttpStatusCode.OK ? $"ords={string.Join(",", split.Ords)}" : split.Body)} (expect 0,2 if supported)"); + + // Verdict is logged verbatim above; assert only the control (harness soundness) hard, and record whether + // ANY category form worked so the report can state PASS/FAIL definitively. + ctrl.Ords.Should().Equal(new[] { 0, 1, 2, 3 }, "CONTROL: schema-name filter works — the store/index harness is sound"); + var categoryWorks = (startsWith.Status == HttpStatusCode.OK && startsWith.Ords.SequenceEqual(new[] { 0, 2 })) + || (split.Status == HttpStatusCode.OK && split.Ords.SequenceEqual(new[] { 0, 2 })); + output.WriteLine($"SPIKE-12 VERDICT: category/stream selection expressible = {categoryWorks} " + + "(if false, EnsureFieldPartitionedIndexAsync must enumerate the category's event types)"); + } + + // ===================================================================================================== + // SPIKE-9 — DIRECT-SUBSCRIBE DEFINITIVE. The whole-index stream "$idx-user-" is NOT directly + // consumable: SubscribeToStream and ReadStream both yield nothing (resolveLinkTos true AND false). The ONLY + // documented recipe is the filtered-$all prefix. This nails "can you subscribe to the index like any stream?". + // ===================================================================================================== + [Fact] + public async Task Spike9_direct_subscribe_and_read_on_index_stream_yield_nothing() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s9"); + await client.AppendToStreamAsync("S9-1", StreamState.NoStream, + [Bodied("UdiA", 0, "x"), Bodied("UdiB", 1, "y"), Bodied("UdiA", 2, "x")]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "s9-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 3, TimeSpan.FromSeconds(30)); + var stream = UserDefinedIndexSource.IndexStream(spec.Name); + + // Sanity: the DOCUMENTED path (filtered $all prefix) DOES work — proves the index has data to find. + var viaFilter = await ReadPrefixOrdsAsync(client, stream); + output.WriteLine($"SPIKE-9 filtered-$all prefix (documented recipe) = {string.Join(",", viaFilter)} (expect 0,1,2)"); + viaFilter.Should().Equal(new[] { 0, 1, 2 }, "the documented filtered-$all recipe must work"); + + // ReadStream x resolveLinkTos matrix. + foreach (var rlt in new[] { true, false }) + { + var res = client.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: rlt); + var state = await res.ReadState; + var n = state == ReadState.StreamNotFound ? 0 : await res.CountAsync(); + output.WriteLine($"SPIKE-9 ReadStream('{stream}', resolveLinkTos={rlt}) -> ReadState={state}, events={n}"); + n.Should().Be(0, $"direct ReadStream on the index stream yields nothing (resolveLinkTos={rlt})"); + } + + // SubscribeToStream x resolveLinkTos matrix — bounded wait, then confirm nothing arrived. + foreach (var rlt in new[] { true, false }) + { + var got = new ConcurrentQueue(); + string? err = null; + using var cts = new CancellationTokenSource(); + var pump = Task.Run(async () => + { + try + { + await using var sub = client.SubscribeToStream(stream, FromStream.Start, resolveLinkTos: rlt, + cancellationToken: cts.Token); + await foreach (var m in sub.Messages.WithCancellation(cts.Token)) + if (m is StreamMessage.Event) got.Enqueue(1); + } + catch (OperationCanceledException) { } + catch (Exception ex) { err = $"{ex.GetType().Name}: {ex.Message}"; } + }); + await WaitUntil(() => Task.FromResult(got.Count >= 3), TimeSpan.FromSeconds(8)); + cts.Cancel(); await pump; + output.WriteLine($"SPIKE-9 SubscribeToStream('{stream}', resolveLinkTos={rlt}) -> events={got.Count}, err={err ?? ""}"); + got.Count.Should().Be(0, $"direct SubscribeToStream on the index stream yields nothing (resolveLinkTos={rlt})"); + } + } + + // ===================================================================================================== + // SPIKE-10 — CaughtUp signal + no-count-gate on the WORKING filtered-$all catch-up path. + // ===================================================================================================== + [Fact] + public async Task Spike10_caughtup_fires_and_live_path_needs_no_count_gate() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s10"); + await client.AppendToStreamAsync("S10-1", StreamState.NoStream, + [Bodied("UdiA", 0, "x"), Bodied("UdiB", 1, "y"), Bodied("UdiA", 2, "x")]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "s10-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 3, TimeSpan.FromSeconds(30)); + var stream = UserDefinedIndexSource.IndexStream(spec.Name); + + var received = new ConcurrentQueue(); + var caughtUpAfter = -1; // received.Count at the moment CaughtUp fired + var caughtUpFired = false; + using var cts = new CancellationTokenSource(); + + // NOTE: NO WaitUntilReadyAsync / count precondition on this LIVE path — just subscribe + tail. + var pump = Task.Run(async () => + { + var filter = new SubscriptionFilterOptions(StreamFilter.Prefix(stream)); + await using var sub = client.SubscribeToAll(FromAll.Start, resolveLinkTos: true, filterOptions: filter, + cancellationToken: cts.Token); + try + { + await foreach (var m in sub.Messages.WithCancellation(cts.Token)) + { + switch (m) + { + case StreamMessage.Event(var e): + received.Enqueue(JsonNode.Parse(e.Event.Data.Span)?["Ord"]?.GetValue() ?? -1); + break; + case StreamMessage.CaughtUp: + caughtUpFired = true; + caughtUpAfter = received.Count; + break; + } + } + } + catch (OperationCanceledException) { } + }); + + // History (0,1,2) then wait for CaughtUp. + await WaitUntil(() => Task.FromResult(received.Count >= 3), TimeSpan.FromSeconds(30)); + var sawCaughtUp = await WaitUntil(() => Task.FromResult(caughtUpFired), TimeSpan.FromSeconds(15)); + output.WriteLine($"SPIKE-10 CaughtUp fired={caughtUpFired} afterN={caughtUpAfter}; history={string.Join(",", received)}"); + + // Live append AFTER caught-up — must tail with no count gate. + await client.AppendToStreamAsync("S10-2", StreamState.NoStream, [Bodied("UdiB", 3, "y"), Bodied("UdiA", 4, "x")]); + var tailed = await WaitUntil(() => Task.FromResult(received.Count >= 5), TimeSpan.FromSeconds(30)); + output.WriteLine($"SPIKE-10 live tail (no count-gate) got={tailed}; all={string.Join(",", received)}"); + + cts.Cancel(); await pump; + + sawCaughtUp.Should().BeTrue("StreamMessage.CaughtUp must fire so a read model can mark itself ready"); + caughtUpAfter.Should().BeGreaterThanOrEqualTo(3, "CaughtUp should fire at/after the full history is delivered"); + tailed.Should().BeTrue("the live subscription tails new appends with NO count-based readiness gate"); + received.Should().Equal(Enumerable.Range(0, 5), "history + live in commit order"); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/IncidentReplayIntegrationTests.cs b/src/MicroPlumberd.Migration.Tests/IncidentReplayIntegrationTests.cs new file mode 100644 index 0000000..fe33495 --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/IncidentReplayIntegrationTests.cs @@ -0,0 +1,310 @@ +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Testing; +using MicroPlumberd.Tests.App.Domain; +using MicroPlumberd.Tests.App.Infrastructure; +using Xunit; + +namespace MicroPlumberd.Migration.Tests; + +/// +/// The definitive proof of the SKIP design against the real incident, using a real MicroPlumberd aggregate +/// () and a real [OutputStream] read model (, which +/// reads the merge stream FooModel_v1) — the OffersReadModel analog. +/// +/// Flow: seed source, build the merge stream via the join projection, HARD-TOMBSTONE one aggregate stream +/// (the wedge), migrate (DropStream the tombstoned one; the copy SKIPS the merge-stream $> links), then +/// boot a FRESH dest plumber and subscribe the read model. The app re-creates the linkTo projection, which +/// regenerates FooModel_v1 from the MIGRATED aggregate streams. Assert: the read model replays to a correct +/// NON-EMPTY projection (survivor present, dropped one absent), no NRE, and the regenerated merge stream has +/// NO duplicate links. +/// +[Trait("Category", "Integration")] +public class IncidentReplayIntegrationTests +{ + private sealed class Server : IAsyncDisposable + { + public required EventStoreServer Es { get; init; } + public required KurrentDBClient Client { get; init; } + public required KurrentDBProjectionManagementClient Projections { get; init; } + public required IPlumber Plumber { get; init; } + public string ConnectionString => Es.HttpUrl.ToString(); + + public static async Task StartAsync(string tag) + { + var es = EventStoreServer.Create($"mp-inc-{tag}-{Guid.NewGuid():N}"); + await es.StartInDocker(inMemory: true); + var settings = es.GetEventStoreSettings(); + return new Server + { + Es = es, + Client = new KurrentDBClient(settings), + Projections = new KurrentDBProjectionManagementClient(settings), + Plumber = global::MicroPlumberd.Plumber.Create(settings) + }; + } + + public async ValueTask DisposeAsync() => await Es.DisposeAsync(); + } + + private static ProjectionCopyContext ProjectionCopy(Server src, Server dst) => new() + { + SourceProjections = src.Projections, + DestProjections = dst.Projections, + SourceConnectionString = src.ConnectionString, + PerEventPaceTimeout = TimeSpan.FromSeconds(30), + DrainTimeout = TimeSpan.FromSeconds(60) + }; + + // The aggregate stream for an id is "-"; discover it empirically rather than hard-coding + // the convention (the stream is the only non-$ stream whose name ends with the id). + private static async Task FindAggregateStreamAsync(KurrentDBClient c, Guid id) + { + var suffix = id.ToString(); + await foreach (var re in c.ReadAllAsync(Direction.Forwards, Position.Start)) + { + var er = re.Event; + if (er is null) continue; + var s = er.EventStreamId; + if (s.Length > 0 && s[0] != '$' && s.EndsWith(suffix, StringComparison.OrdinalIgnoreCase)) + return s; + } + throw new InvalidOperationException($"No aggregate stream found for {id}."); + } + + private static async Task CountStreamAsync(KurrentDBClient c, string stream) + { + var res = c.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: false); + if (await res.ReadState == ReadState.StreamNotFound) return 0; + return await res.LongCountAsync(); + } + + // Reads the join/output stream resolving links, returning each target event's Name in order. A null + // resolved event (the poison) is surfaced as null so the test can assert there are none. + private static async Task> ReadJoinNamesAsync(KurrentDBClient c, string stream) + { + var res = c.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: true); + var names = new List(); + if (await res.ReadState == ReadState.StreamNotFound) return names; + await foreach (var re in res) + { + if (re.Event is null) { names.Add(null); continue; } // dead link — must not happen on dest + var node = System.Text.Json.Nodes.JsonNode.Parse(re.Event.Data.Span); + names.Add(node?["Name"]?.GetValue()); + } + return names; + } + + [Fact] + public async Task Read_model_replays_rebuilt_store_to_correct_ordered_projection_without_NRE_or_duplicate_links() + { + await using var src = await Server.StartAsync("src"); + await using var dst = await Server.StartAsync("dst"); + + // 1. Seed source in a defined order across 4 aggregate streams — the 3rd is dropped, so the + // survivors' relative order (live1, live2, live3) is meaningful to assert after regeneration. + var live1 = Guid.NewGuid(); + var live2 = Guid.NewGuid(); + var deadId = Guid.NewGuid(); + var live3 = Guid.NewGuid(); + await src.Plumber.SaveNew(FooAggregate.Open("live1", live1)); + await src.Plumber.SaveNew(FooAggregate.Open("live2", live2)); + await src.Plumber.SaveNew(FooAggregate.Open("dead", deadId)); + await src.Plumber.SaveNew(FooAggregate.Open("live3", live3)); + + var deadStream = await FindAggregateStreamAsync(src.Client, deadId); + + // 2. Build the [OutputStream] JOIN projection on source (fromStreams(['$et-FooCreated',…]).linkTo + // ('FooModel_v1')) — the ACTUAL ERP mechanism (TryCreateJoinProjection). Wait for all 4 links. + await src.Plumber.TryCreateJoinProjection(); + await WaitUntil(async () => await CountStreamAsync(src.Client, "FooModel_v1") >= 4, TimeSpan.FromSeconds(30)); + + // 3. Hard-tombstone the dead aggregate stream — this is the wedge: FooModel_v1 now has a link to a + // dead event, which NREs when resolved during replay on the SOURCE. + await src.Client.TombstoneAsync(deadStream, StreamState.Any); + + // 4. Migrate source -> fresh dest, dropping the tombstoned stream. The copy skips the $> merge links. + var migration = new DropStreamMigration("0001_drop_dead_offer", deadStream); + var result = await new MigrationRunner().RunAsync(src.Client, dst.Client, [migration], dryRun: false); + + result.Copy.LinkEventsSkipped.Should().BeGreaterThan(0, "FooModel_v1 join links must be skipped, not copied"); + result.Verification!.AllOk.Should().BeTrue(result.Verification.Format()); + (await CountStreamAsync(dst.Client, deadStream)).Should().Be(0, "the tombstoned aggregate stream is dropped"); + + // 5. Boot the read model on the FRESH dest. SubscribeEventHandler re-registers the JOIN projection, + // which REPOPULATES FooModel_v1 from the migrated (dead-free) $et-FooCreated stream; model replays. + var model = new FooModel(new InMemoryAssertionDb()); + await dst.Plumber.SubscribeEventHandler(model); + + // Bounded stability wait — the model must reach 3 events AND hold there across consecutive reads + // (no late dead/duplicate event), instead of a wall-clock sleep. A real dead-link NRE would leave + // the model short; on timeout we surface the ROOT CAUSE (the resolved join links, showing any dead + // link) so it fails LOUD as an NRE rather than as a silent empty-projection timeout. + var stable = await WaitUntilStable(() => model.AssertionDb.Index.Count, target: 3, TimeSpan.FromSeconds(30)); + if (!stable) + { + var diag = await ReadJoinNamesAsync(dst.Client, "FooModel_v1"); + throw new Xunit.Sdk.XunitException( + $"Read model did not replay to a stable 3 events (got {model.AssertionDb.Index.Count}). " + + $"FooModel_v1 resolved to [{string.Join(", ", diag.Select(n => n ?? ""))}] " + + "— a entry means a join link resolved to a deleted event (the incident)."); + } + + // 6. Read model replays to a correct NON-EMPTY projection, in order, no NRE, dropped offer absent. + var modelNames = model.AssertionDb.Index.Values.Select(i => i.Event).OfType() + .Select(e => e.Name).ToList(); + modelNames.Should().Equal("live1", "live2", "live3"); + modelNames.Should().NotContain("dead"); + + // 7. The output stream is REBUILT correctly by the typical projection: same survivors, in order, + // every link resolving (no dead link), and NO duplicate links (proves skip-not-rebuild was right — + // a manual rebuild would have doubled these once the projection re-ran). + var joinNames = await ReadJoinNamesAsync(dst.Client, "FooModel_v1"); + joinNames.Should().NotContainNulls("no join link may resolve to a dead event"); + joinNames.Should().Equal("live1", "live2", "live3"); + (await CountStreamAsync(dst.Client, "FooModel_v1")).Should().Be(3, "exactly one link per survivor — no duplicates"); + } + + // Seeds one aggregate 'a' and one 'b', each FooCreated then FooRefined, in INTERLEAVED $all arrival order + // [a-created, a-refined, b-created, b-refined], then creates the [OutputStream] JOIN projection on source. + private static async Task SeedInterleavedAsync(Server src, Guid a, Guid b) + { + await src.Plumber.SaveNew(FooAggregate.Open("a-created", a)); // FooCreated(a) + var aggA = await src.Plumber.Get(a); + aggA.Refine("a-refined"); await src.Plumber.SaveChanges(aggA); // FooRefined(a) + await src.Plumber.SaveNew(FooAggregate.Open("b-created", b)); // FooCreated(b) + var aggB = await src.Plumber.Get(b); + aggB.Refine("b-refined"); await src.Plumber.SaveChanges(aggB); // FooRefined(b) + await src.Plumber.TryCreateJoinProjection(); + await WaitUntil(async () => await CountStreamAsync(src.Client, "FooModel_v1") >= 4, TimeSpan.FromSeconds(30)); + } + + // THE DETERMINISM GATE. The paced PROJECTION COPY (pre-create the fromStreams join projection on the empty + // dest, then feed events one at a time waiting for each link before the next) must rebuild the merge stream + // in ARRIVAL order EVERY time. The earlier end-only-drain version was 7/8 (a backlog type-clustered once); + // this runs the interleaved recovery 15× and requires 15/15 — a single cluster fails it loudly. + [Fact] + public async Task Paced_projection_copy_preserves_interleaved_order_deterministically_15x() + { + await using var src = await Server.StartAsync("detsrc"); + await SeedInterleavedAsync(src, Guid.NewGuid(), Guid.NewGuid()); + var expected = new[] { "a-created", "a-refined", "b-created", "b-refined" }; + + var observed = new List(); + for (var i = 0; i < 15; i++) + { + await using var dst = await Server.StartAsync($"detdst{i}"); + await new MigrationRunner().RunAsync(src.Client, dst.Client, Array.Empty(), dryRun: false, + ProjectionCopy(src, dst)); + await WaitUntil(async () => await CountStreamAsync(dst.Client, "FooModel_v1") >= 4, TimeSpan.FromSeconds(30)); + var order = await ReadJoinNamesAsync(dst.Client, "FooModel_v1"); + observed.Add(string.Join(",", order.Select(n => n ?? ""))); + } + + observed.Should().OnlyContain(o => o == string.Join(",", expected), + "paced projection copy must be DETERMINISTIC — 15/15 in arrival order. Observed per run: [{0}]", + string.Join(" | ", observed)); + } + + // THE END-TO-END RECOVERY PROOF (paced projection copy). The real :5081 fix: a tombstoned aggregate stream + // poisons the merge stream (dead link → NRE). The migration drops it, PRE-CREATES the join projection on the + // empty dest, and PACES the copy so the projection rebuilds FooModel_v1 in commit order as the sole writer; + // the app boots and NO-OPs (mp_query_hash match). Interleaved multi-type fixture makes clustering fail loudly. + [Fact] + public async Task Paced_projection_copy_recovers_tombstoned_store_in_commit_order_without_NRE() + { + await using var src = await Server.StartAsync("recsrc"); + await using var dst = await Server.StartAsync("recdst"); + + var a = Guid.NewGuid(); + var dead = Guid.NewGuid(); + var b = Guid.NewGuid(); + await src.Plumber.SaveNew(FooAggregate.Open("a-created", a)); + var aggA = await src.Plumber.Get(a); + aggA.Refine("a-refined"); await src.Plumber.SaveChanges(aggA); + await src.Plumber.SaveNew(FooAggregate.Open("dead", dead)); // poisons the merge stream, then dropped + await src.Plumber.SaveNew(FooAggregate.Open("b-created", b)); + var aggB = await src.Plumber.Get(b); + aggB.Refine("b-refined"); await src.Plumber.SaveChanges(aggB); + + var deadStream = await FindAggregateStreamAsync(src.Client, dead); + await src.Plumber.TryCreateJoinProjection(); + await WaitUntil(async () => await CountStreamAsync(src.Client, "FooModel_v1") >= 5, TimeSpan.FromSeconds(30)); + await src.Client.TombstoneAsync(deadStream, StreamState.Any); // the wedge (dead link in FooModel_v1) + + // Migrate: drop the tombstoned stream + PACED projection copy → the pre-created join projection rebuilds + // FooModel_v1 in commit order from the migrated (dead-free) store. + var migration = new DropStreamMigration("0001_drop_dead_offer", deadStream); + var result = await new MigrationRunner().RunAsync(src.Client, dst.Client, [migration], dryRun: false, + ProjectionCopy(src, dst)); + + result.CopiedProjections.Should().Contain("FooModel_v1", "the app's join projection is pre-created on the dest"); + result.Copy.LinkEventsSkipped.Should().BeGreaterThan(0, "source FooModel_v1 $> links are skipped, not copied"); + result.Verification!.AllOk.Should().BeTrue(result.Verification.Format()); + (await CountStreamAsync(dst.Client, deadStream)).Should().Be(0, "the tombstoned stream is dropped"); + + // Merge stream built by the paced projection: survivors only, commit order, no dead link, no duplicates. + await WaitUntil(async () => await CountStreamAsync(dst.Client, "FooModel_v1") >= 4, TimeSpan.FromSeconds(30)); + var joinNames = await ReadJoinNamesAsync(dst.Client, "FooModel_v1"); + joinNames.Should().NotContainNulls("no join link may resolve to a dead event"); + joinNames.Should().Equal(new[] { "a-created", "a-refined", "b-created", "b-refined" }, + "paced projection copy must preserve commit order, not type-cluster"); + joinNames.Should().NotContain("dead"); + (await CountStreamAsync(dst.Client, "FooModel_v1")).Should().Be(4, "one link per survivor — no duplicates"); + + // The read model replays clean over the pre-built dest — NON-EMPTY, in order, NO NRE. + var model = new FooModel(new InMemoryAssertionDb()); + await dst.Plumber.SubscribeEventHandler(model); + var stable = await WaitUntilStable(() => model.AssertionDb.Index.Count, target: 4, TimeSpan.FromSeconds(30)); + if (!stable) + { + var diag = await ReadJoinNamesAsync(dst.Client, "FooModel_v1"); + throw new Xunit.Sdk.XunitException( + $"Read model did not replay to a stable 4 events (got {model.AssertionDb.Index.Count}). " + + $"FooModel_v1 resolved to [{string.Join(", ", diag.Select(n => n ?? ""))}]."); + } + + // Simulated APP BOOT: fresh plumber re-registers the join projection → NO-OP (mp_query_hash match), so + // the merge stream is left untouched (no restart / re-link / reorder / duplicate). + var countBefore = await CountStreamAsync(dst.Client, "FooModel_v1"); + var bootPlumber = global::MicroPlumberd.Plumber.Create(dst.Es.GetEventStoreSettings()); + (await bootPlumber.TryCreateJoinProjection()).Should().BeFalse("mp_query_hash matches → app no-ops"); + await Task.Delay(1000); + (await ReadJoinNamesAsync(dst.Client, "FooModel_v1")).Should().Equal(new[] { "a-created", "a-refined", "b-created", "b-refined" }); + (await CountStreamAsync(dst.Client, "FooModel_v1")).Should().Be(countBefore, "link count stable across the no-op boot"); + } + + private static async Task WaitUntil(Func> condition, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + if (await condition()) return; + await Task.Delay(500); + } + } + + // Polls value until it reaches target AND holds that value across two consecutive reads (stable), or + // times out. Returns whether stability at the target was observed. + private static async Task WaitUntilStable(Func value, int target, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + var prev = -1; + while (DateTime.UtcNow < deadline) + { + var cur = value(); + if (cur >= target && cur == prev) return true; + prev = cur; + await Task.Delay(500); + } + return false; + } + + /// A raw DropStream migration over an exact stream name (discovered at runtime). + private sealed class DropStreamMigration(string id, string stream) : Migration + { + public override string Id => id; + public override void Migrate(IMigrationBuilder b) => b.DropStream(stream); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/LiveIndexSubscriptionSpikes.cs b/src/MicroPlumberd.Migration.Tests/LiveIndexSubscriptionSpikes.cs new file mode 100644 index 0000000..befd0cd --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/LiveIndexSubscriptionSpikes.cs @@ -0,0 +1,737 @@ +using System.Collections.Concurrent; +using System.Net; +using System.Text; +using System.Text.Json.Nodes; +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Testing; +using Xunit; +using Xunit.Abstractions; + +namespace MicroPlumberd.Migration.Tests; + +/// +/// SPIKE — throwaway empirical feasibility investigation (NOT production). Answers ONE yes/no question against a +/// REAL KurrentDB 26.1 (MicroPlumberd.Testing Docker server, NEVER Testcontainers): can a KurrentDB 26.1 +/// USER-DEFINED INDEX ($idx-user-<name>) back a LIVE read-model merge stream — i.e. replace a +/// fromStreams([$et-A,$et-B]).linkTo(out) join projection — such that a read model can SUBSCRIBE +/// (catch-up over history THEN tail live) and receive events in COMMIT order as new ones are appended? +/// +/// The existing only does a BOUNDED historical read; these spikes probe the +/// LIVE behaviour. Each test captures ACTUAL observed evidence (via ) and asserts. +/// +/// SPIKE-1: does the index LIVE-UPDATE and stay COMMIT-ordered when more interleaved events are appended AFTER it +/// is already built (bounded re-read). +/// SPIKE-2: can KurrentDB.Client SUBSCRIBE (catch-up→live) to the index and RECEIVE post-subscription appends +/// through the same open subscription, in commit order — via SubscribeToStream on the index stream AND +/// via filtered SubscribeToAll (the ReadAsync analog). Reports exactly which API works. +/// SPIKE-3: can a position in the index stream be recorded, the subscription stopped, and RESUMED from it (as a +/// restarting read model must) — receiving ONLY the new events. +/// SPIKE-4: can a user-defined index express BODY-PROPERTY routing (the linkTo('cat-'+e.body.SomeId) +/// lookup-projection case), or is it limited to schema/stream filtering. +/// +[Trait("Category", "Integration")] +public class LiveIndexSubscriptionSpikes(ITestOutputHelper output) +{ + private sealed class Store : IAsyncDisposable + { + private readonly List _servers = new(); + + public async Task<(KurrentDBClient Client, string Conn)> NewAsync(string tag) + { + var es = EventStoreServer.Create($"mp-live-{tag}-{Guid.NewGuid():N}"); + _servers.Add(es); + await es.StartInDocker(inMemory: true); + var s = es.GetEventStoreSettings(); + return (new KurrentDBClient(s), es.HttpUrl.ToString()); + } + + // Variant that also hands back a persistent-subscriptions client (for the persistent-sub spike). + public async Task<(KurrentDBClient Client, KurrentDBPersistentSubscriptionsClient Persistent, string Conn)> NewWithPsAsync(string tag) + { + var es = EventStoreServer.Create($"mp-live-{tag}-{Guid.NewGuid():N}"); + _servers.Add(es); + await es.StartInDocker(inMemory: true); + var s = es.GetEventStoreSettings(); + return (new KurrentDBClient(s), new KurrentDBPersistentSubscriptionsClient(s), es.HttpUrl.ToString()); + } + + public async ValueTask DisposeAsync() + { + foreach (var s in _servers) await s.DisposeAsync(); + } + } + + private static EventData Typed(string type, int ord) => + new(Uuid.NewUuid(), type, Encoding.UTF8.GetBytes($"{{\"Ord\":{ord}}}")); + + private static EventData Bodied(string type, int ord, string someId) => + new(Uuid.NewUuid(), type, Encoding.UTF8.GetBytes($"{{\"Ord\":{ord},\"SomeId\":\"{someId}\"}}")); + + private static async Task> ReadOrdsAsync(UserDefinedIndexSource src, string name, CancellationToken ct = default) + { + var ords = new List(); + await foreach (var e in src.ReadAsync(name, ct)) + ords.Add(e.Data?["Ord"]?.GetValue() ?? -1); + return ords; + } + + private static async Task WaitUntil(Func> condition, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + if (await condition()) return true; + await Task.Delay(200); + } + return await condition(); + } + + // ===================================================================================================== + // SPIKE-1 — LIVE TAIL ORDERING (the load-bearing test): after the index is built over an initial + // interleaved batch, append MORE interleaved events and verify the index picks them up (start:true tails) + // AND keeps them in commit order 0..N-1 — NOT re-clustered, NOT stale. + // ===================================================================================================== + [Fact] + public async Task Spike1_index_live_updates_and_stays_commit_ordered_when_more_events_appended() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s1"); + + // Initial interleaved batch A0,B1,A2,B3,A4. + await client.AppendToStreamAsync("S1-a", StreamState.NoStream, + [Typed("UdiA", 0), Typed("UdiB", 1), Typed("UdiA", 2), Typed("UdiB", 3), Typed("UdiA", 4)]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "s1-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 5, TimeSpan.FromSeconds(30)); + + var before = await ReadOrdsAsync(src, spec.Name); + output.WriteLine("SPIKE-1 before live-append: " + string.Join(",", before)); + before.Should().Equal(new[] { 0, 1, 2, 3, 4 }); + + // NOW append MORE interleaved events to a DIFFERENT stream (later commit positions 5..9). + await client.AppendToStreamAsync("S1-b", StreamState.NoStream, + [Typed("UdiB", 5), Typed("UdiA", 6), Typed("UdiB", 7), Typed("UdiA", 8), Typed("UdiB", 9)]); + + // Poll the index (bounded re-read) until it reflects all 10, capturing whether it live-updates at all. + List after = new(); + var reached = await WaitUntil(async () => + { + after = await ReadOrdsAsync(src, spec.Name); + return after.Count >= 10; + }, TimeSpan.FromSeconds(30)); + + output.WriteLine($"SPIKE-1 after live-append (reached10={reached}): " + string.Join(",", after)); + + reached.Should().BeTrue("a start:true user-defined index must keep indexing events appended AFTER creation"); + after.Should().Equal(Enumerable.Range(0, 10), + "newly-appended interleaved events must land in the index in COMMIT order 0..9, not re-clustered"); + } + + // ===================================================================================================== + // SPIKE-2 — SUBSCRIBE (catch-up → live), not just read. Open a subscription BEFORE the live appends and + // prove the post-subscription appends arrive through the SAME open subscription, in commit order. + // Tries BOTH mechanisms and reports which works. + // ===================================================================================================== + + // 2a: SubscribeToStream on the index stream itself ($idx-user-). + // FINDING RECORDED (feasibility §7 / design R1): a direct SubscribeToStream on $idx-user-* silently delivers ZERO + // events on KurrentDB 26.1 — this spike stays red by construction and its job is done; the loud guard in + // SubscriptionRunnerState.Subscribe() and Spike2b (filtered $all) are the live tests. Skipped, not deleted. + [Fact(Skip = "SPIKE-2a is a recorded negative finding: SubscribeToStream on $idx-user-* delivers zero events (design R1). See Spike2b.")] + public async Task Spike2a_subscribe_to_index_stream_catches_up_then_receives_live_appends_in_order() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s2a"); + + await client.AppendToStreamAsync("S2a-a", StreamState.NoStream, + [Typed("UdiA", 0), Typed("UdiB", 1), Typed("UdiA", 2)]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "s2a-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 3, TimeSpan.FromSeconds(30)); + + var indexStream = UserDefinedIndexSource.IndexStream(spec.Name); // "$idx-user-s2a-merge" + var received = new ConcurrentQueue(); + using var cts = new CancellationTokenSource(); + string? subscribeError = null; + + var pump = Task.Run(async () => + { + try + { + await using var sub = client.SubscribeToStream(indexStream, FromStream.Start, resolveLinkTos: true, + cancellationToken: cts.Token); + await foreach (var m in sub.Messages.WithCancellation(cts.Token)) + { + if (m is StreamMessage.Event(var e)) + { + var ord = JsonNode.Parse(e.Event.Data.Span)?["Ord"]?.GetValue() ?? -1; + received.Enqueue(ord); + } + } + } + catch (OperationCanceledException) { /* expected on stop */ } + catch (Exception ex) { subscribeError = $"{ex.GetType().Name}: {ex.Message}"; } + }); + + // Catch-up: the 3 historical events should arrive. + var caughtUp = await WaitUntil(() => Task.FromResult(received.Count >= 3), TimeSpan.FromSeconds(30)); + output.WriteLine($"SPIKE-2a subscribe error: {subscribeError ?? ""}"); + output.WriteLine($"SPIKE-2a after catch-up (caughtUp={caughtUp}): " + string.Join(",", received)); + + if (subscribeError is not null) + { + // CRITICAL finding path — surface the exact error and stop (cannot subscribe to the index stream). + cts.Cancel(); + subscribeError.Should().BeNull( + "SUBSCRIBING to $idx-user- via SubscribeToStream must work for an index-backed live read model"); + } + + caughtUp.Should().BeTrue("the subscription must catch up over the index's history"); + + // LIVE: append AFTER the subscription is open; the SAME subscription must push them. + await client.AppendToStreamAsync("S2a-b", StreamState.NoStream, [Typed("UdiB", 3), Typed("UdiA", 4)]); + var gotLive = await WaitUntil(() => Task.FromResult(received.Count >= 5), TimeSpan.FromSeconds(30)); + + cts.Cancel(); + await pump; + + output.WriteLine($"SPIKE-2a after live-append (gotLive={gotLive}): " + string.Join(",", received)); + gotLive.Should().BeTrue("post-subscription appends must be PUSHED to the open index-stream subscription"); + received.Should().Equal(Enumerable.Range(0, 5), + "the subscription must deliver history + live events in commit order 0..4"); + } + + // 2b: filtered SubscribeToAll over StreamFilter.Prefix($idx-user-) — the subscription analog of the + // filtered $all read ReadAsync already uses. This is the most likely real mechanism. + [Fact] + public async Task Spike2b_filtered_subscribe_to_all_catches_up_then_receives_live_appends_in_order() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s2b"); + + await client.AppendToStreamAsync("S2b-a", StreamState.NoStream, + [Typed("UdiA", 0), Typed("UdiB", 1), Typed("UdiA", 2)]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "s2b-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 3, TimeSpan.FromSeconds(30)); + + var indexStream = UserDefinedIndexSource.IndexStream(spec.Name); + var received = new ConcurrentQueue(); + using var cts = new CancellationTokenSource(); + string? subscribeError = null; + + var pump = Task.Run(async () => + { + try + { + var filter = new SubscriptionFilterOptions(StreamFilter.Prefix(indexStream)); + await using var sub = client.SubscribeToAll(FromAll.Start, resolveLinkTos: true, + filterOptions: filter, cancellationToken: cts.Token); + await foreach (var m in sub.Messages.WithCancellation(cts.Token)) + { + if (m is StreamMessage.Event(var e)) + { + var ord = JsonNode.Parse(e.Event.Data.Span)?["Ord"]?.GetValue() ?? -1; + received.Enqueue(ord); + } + } + } + catch (OperationCanceledException) { /* expected on stop */ } + catch (Exception ex) { subscribeError = $"{ex.GetType().Name}: {ex.Message}"; } + }); + + var caughtUp = await WaitUntil(() => Task.FromResult(received.Count >= 3), TimeSpan.FromSeconds(30)); + output.WriteLine($"SPIKE-2b subscribe error: {subscribeError ?? ""}"); + output.WriteLine($"SPIKE-2b after catch-up (caughtUp={caughtUp}): " + string.Join(",", received)); + + caughtUp.Should().BeTrue("filtered $all subscription must catch up over the index's history"); + + await client.AppendToStreamAsync("S2b-b", StreamState.NoStream, [Typed("UdiB", 3), Typed("UdiA", 4)]); + var gotLive = await WaitUntil(() => Task.FromResult(received.Count >= 5), TimeSpan.FromSeconds(30)); + + cts.Cancel(); + await pump; + + output.WriteLine($"SPIKE-2b subscribe error (final): {subscribeError ?? ""}"); + output.WriteLine($"SPIKE-2b after live-append (gotLive={gotLive}): " + string.Join(",", received)); + subscribeError.Should().BeNull("filtered $all subscription must not fault"); + gotLive.Should().BeTrue("post-subscription appends must be PUSHED to the open filtered-$all subscription"); + received.Should().Equal(Enumerable.Range(0, 5), + "the filtered subscription must deliver history + live events in commit order 0..4"); + } + + // ===================================================================================================== + // SPIKE-3 — CHECKPOINT / RESUME. Record a position, stop, append more, resume from the recorded position, + // and receive ONLY the new events (as a restarting read model must). Uses SubscribeToAll+filter and + // resumes from the recorded $all Position (the mechanism that mirrors ReadAsync). If 2a works, the + // index-stream revision path is also viable; this test uses the $all Position path. + // ===================================================================================================== + [Fact] + public async Task Spike3_can_record_position_stop_and_resume_receiving_only_new_events() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s3"); + + await client.AppendToStreamAsync("S3-a", StreamState.NoStream, + [Typed("UdiA", 0), Typed("UdiB", 1), Typed("UdiA", 2)]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "s3-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 3, TimeSpan.FromSeconds(30)); + + var indexStream = UserDefinedIndexSource.IndexStream(spec.Name); + var filter = new SubscriptionFilterOptions(StreamFilter.Prefix(indexStream)); + + // Pass 1: consume the 3 historical events, recording the LAST link's $all Position as the checkpoint. + var firstPass = new List(); + Position? checkpoint = null; + using (var cts1 = new CancellationTokenSource()) + { + var pump1 = Task.Run(async () => + { + await using var sub = client.SubscribeToAll(FromAll.Start, resolveLinkTos: true, + filterOptions: filter, cancellationToken: cts1.Token); + try + { + await foreach (var m in sub.Messages.WithCancellation(cts1.Token)) + if (m is StreamMessage.Event(var e)) + { + firstPass.Add(JsonNode.Parse(e.Event.Data.Span)?["Ord"]?.GetValue() ?? -1); + if (e.OriginalPosition is { } p) checkpoint = p; // link's $all position — the resume point + } + } + catch (OperationCanceledException) { } + }); + await WaitUntil(() => Task.FromResult(firstPass.Count >= 3), TimeSpan.FromSeconds(30)); + cts1.Cancel(); + await pump1; + } + + output.WriteLine($"SPIKE-3 pass1: {string.Join(",", firstPass)} checkpoint={checkpoint?.ToString() ?? ""}"); + firstPass.Should().Equal(new[] { 0, 1, 2 }); + checkpoint.Should().NotBeNull("a resumable $all position must be recoverable from the index link event"); + + // Append MORE while "stopped". + await client.AppendToStreamAsync("S3-b", StreamState.NoStream, [Typed("UdiB", 3), Typed("UdiA", 4)]); + + // Pass 2: RESUME from the recorded checkpoint (FromAll.After) — must receive ONLY 3,4 (not 0,1,2 again). + var secondPass = new ConcurrentQueue(); + using (var cts2 = new CancellationTokenSource()) + { + var pump2 = Task.Run(async () => + { + await using var sub = client.SubscribeToAll(FromAll.After(checkpoint!.Value), resolveLinkTos: true, + filterOptions: filter, cancellationToken: cts2.Token); + try + { + await foreach (var m in sub.Messages.WithCancellation(cts2.Token)) + if (m is StreamMessage.Event(var e)) + secondPass.Enqueue(JsonNode.Parse(e.Event.Data.Span)?["Ord"]?.GetValue() ?? -1); + } + catch (OperationCanceledException) { } + }); + await WaitUntil(() => Task.FromResult(secondPass.Count >= 2), TimeSpan.FromSeconds(30)); + cts2.Cancel(); + await pump2; + } + + output.WriteLine($"SPIKE-3 pass2 (resumed): {string.Join(",", secondPass)}"); + secondPass.Should().Equal(new[] { 3, 4 }, + "resuming from the recorded position must yield ONLY events after it, in commit order — no replay of 0,1,2"); + } + + // ===================================================================================================== + // SPIKE-4 — BODY-PROPERTY ROUTING (the likely hard limit). The current lookup projection does + // linkTo('cat-' + e.body.SomeId): ONE stream PER distinct SomeId value. Empirically probe whether a + // user-defined index filter can (a) reference the event BODY at all, and note that even if it can, an + // index still produces ONE merged stream — it cannot PARTITION into N per-value streams. + // ===================================================================================================== + [Fact] + public async Task Spike4_body_property_routing_is_not_expressible_as_a_user_defined_index() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s4"); + + // Two SomeId partitions x,y interleaved. A lookup projection would route these to cat-x and cat-y. + await client.AppendToStreamAsync("S4-a", StreamState.NoStream, new[] + { + Bodied("UdiC", 0, "x"), Bodied("UdiC", 1, "y"), Bodied("UdiC", 2, "x"), Bodied("UdiC", 3, "y") + }); + + // Probe: try to POST a user-defined index whose filter references the event BODY (rec.data / rec.body). + // We hit the same /v2/indexes endpoint EnsureAsync uses, but with a body-referencing filter, and capture + // the raw HTTP outcome + whether it indexes the expected subset. + var (httpBase, user, pass) = KurrentHttpEndpoint.Parse(conn); + using var http = KurrentHttpEndpoint.CreateClient(user, pass); + + var probes = new[] + { + ("data", "rec => rec.data.SomeId == \"x\""), + ("body", "rec => rec.body.SomeId == \"x\""), + ("dataloweronly", "rec => rec.data.someid == \"x\""), + // Baseline control: a SCHEMA-name filter with the SAME lowercase name MUST be accepted — proves any + // 400 on the body probes above is about the FILTER, not the (now-valid) index name. + ("schemacontrol", "rec => rec.schema.name == \"UdiC\""), + }; + for (var i = 0; i < probes.Length; i++) + { + var (label, filterExpr) = probes[i]; + var name = $"s4-{label}"; // lowercase alnum + dash only — a valid index name + var url = new Uri(httpBase, $"v2/indexes/{Uri.EscapeDataString(name)}"); + var payload = new JsonObject { ["filter"] = filterExpr, ["start"] = true }; + using var content = new StringContent(payload.ToJsonString(), Encoding.UTF8, "application/json"); + var resp = await http.PostAsync(url, content); + var body = await resp.Content.ReadAsStringAsync(); + output.WriteLine($"SPIKE-4 POST filter [{filterExpr}] -> HTTP {(int)resp.StatusCode} {resp.StatusCode}: {body}"); + + if (resp.IsSuccessStatusCode) + { + // If accepted, check whether it actually indexed only the x-partition (2 events) — i.e. whether the + // engine evaluated the body predicate. Read via filtered $all. + var idxSrc = new UserDefinedIndexSource(client, conn); + await Task.Delay(1500); // let it backfill + var ords = new List(); + try { ords = await ReadOrdsAsync(idxSrc, name); } catch (Exception ex) { output.WriteLine($" read err: {ex.Message}"); } + output.WriteLine($" indexed ords for [{filterExpr}]: {string.Join(",", ords)} (x-partition would be 0,2)"); + } + } + + // The structural point, independent of whether body filtering is accepted: a user-defined index is a + // FILTER producing ONE $idx-user- stream. It has NO 'linkTo(dynamic-name)' — it cannot emit N + // streams keyed by a body value. So the lookup/by-key projection (cat-) is NOT replaceable by an + // index; this assertion documents the empirical conclusion (evidence is in the logged POST outcomes). + output.WriteLine("SPIKE-4 conclusion: a user-defined index cannot PARTITION by body value (no dynamic linkTo); " + + "at best it filters to one merged stream. Lookup-by-key projections are NOT index-replaceable."); + true.Should().BeTrue(); + } + + // ===================================================================================================== + // SPIKE-5 (arch §3 Q4) — FILTER DRIFT / RECREATION COST. Can an existing index's filter be mutated in + // place? If not, is there a DELETE, and what does delete+recreate+backfill cost for a realistic size? + // This becomes the "projection update" replacement for mp_query_hash's disable/update/enable. + // ===================================================================================================== + [Fact] + public async Task Spike5_filter_drift_and_recreation_cost() + { + await using var stores = new Store(); + var (client, conn) = await stores.NewAsync("s5"); + + // A realistic-size store: 5000 UdiA + 5000 UdiB interleaved (10k total). + const int per = 5000; + var expected = StreamState.NoStream; + var total = 0; + for (var offset = 0; offset < per * 2; offset += 1000) + { + var chunk = Enumerable.Range(offset, 1000).Select(i => Typed(i % 2 == 0 ? "UdiA" : "UdiB", i)).ToArray(); + await client.AppendToStreamAsync("S5-1", expected, chunk); + expected = (ulong)(offset + chunk.Length - 1); + total += chunk.Length; + } + + var (httpBase, user, pass) = KurrentHttpEndpoint.Parse(conn); + using var http = KurrentHttpEndpoint.CreateClient(user, pass); + var src = new UserDefinedIndexSource(client, conn); + + // Create index filtering ONLY UdiA (5000 events). + var name = "s5-drift"; + var url = new Uri(httpBase, $"v2/indexes/{name}"); + async Task<(HttpStatusCode, string)> Post(string filter) + { + using var c = new StringContent(new JsonObject { ["filter"] = filter, ["start"] = true }.ToJsonString(), + Encoding.UTF8, "application/json"); + var r = await http.PostAsync(url, c); + return (r.StatusCode, await r.Content.ReadAsStringAsync()); + } + async Task GetFilter() + { + var r = await http.GetAsync(url); + if (!r.IsSuccessStatusCode) return $""; + return JsonNode.Parse(await r.Content.ReadAsStringAsync())?["index"]?["filter"]?.GetValue(); + } + + var onlyA = "rec => rec.schema.name == \"UdiA\""; + var aAndB = "rec => rec.schema.name == \"UdiA\" || rec.schema.name == \"UdiB\""; + + var (createStatus, createBody) = await Post(onlyA); + output.WriteLine($"SPIKE-5 create(onlyA) -> HTTP {(int)createStatus}: {createBody}"); + await src.WaitUntilReadyAsync(name, expectedCount: per, TimeSpan.FromSeconds(120)); + output.WriteLine($"SPIKE-5 filter after create: {await GetFilter()}"); + + // (a) Attempt in-place mutation: POST same name, DIFFERENT filter (A+B). + var (mutStatus, mutBody) = await Post(aAndB); + var filterAfterMutate = await GetFilter(); + output.WriteLine($"SPIKE-5 mutate(A+B) POST -> HTTP {(int)mutStatus}: {mutBody}"); + output.WriteLine($"SPIKE-5 filter after mutate attempt: {filterAfterMutate} (unchanged={filterAfterMutate == onlyA})"); + + // (b) Is there a DELETE? Capture verbatim. + var delResp = await http.DeleteAsync(url); + var delBody = await delResp.Content.ReadAsStringAsync(); + output.WriteLine($"SPIKE-5 DELETE -> HTTP {(int)delResp.StatusCode} {delResp.StatusCode}: {delBody}"); + + // (c) If delete worked, measure delete+recreate+backfill cost for the WIDENED filter (A+B = 10k). + if (delResp.IsSuccessStatusCode) + { + // small settle so the delete is visible + await Task.Delay(500); + var sw = System.Diagnostics.Stopwatch.StartNew(); + var (recStatus, recBody) = await Post(aAndB); + output.WriteLine($"SPIKE-5 recreate(A+B) -> HTTP {(int)recStatus}: {recBody}"); + try + { + await src.WaitUntilReadyAsync(name, expectedCount: total, TimeSpan.FromSeconds(180)); + sw.Stop(); + output.WriteLine($"SPIKE-5 recreate+backfill of {total} events took {sw.ElapsedMilliseconds} ms"); + var widened = await GetFilter(); + output.WriteLine($"SPIKE-5 filter after recreate: {widened}"); + } + catch (Exception ex) + { + sw.Stop(); + output.WriteLine($"SPIKE-5 recreate backfill FAILED after {sw.ElapsedMilliseconds} ms: {ex.Message}"); + } + } + else + { + output.WriteLine("SPIKE-5: no working DELETE — an index cannot be cheaply re-pointed; a filter change " + + "requires a NEW index name (old one lingers). This is the update-cost finding."); + } + + // No hard assert on the timing (informational), but the in-place mutation MUST NOT silently change the + // filter — that's the correctness-relevant part (drift is either rejected or ignored, never a silent swap). + // If the server DID redefine in place, filterAfterMutate would equal aAndB; record whichever happened. + (filterAfterMutate == onlyA || filterAfterMutate == aAndB || filterAfterMutate!.StartsWith(" Typed(i % 2 == 0 ? "UdiA" : "UdiB", i)).ToArray()); + + var src = new UserDefinedIndexSource(client, conn); + const int count = 60; + var failures = new List(); + var sw = System.Diagnostics.Stopwatch.StartNew(); + for (var i = 0; i < count; i++) + { + try + { + await src.EnsureAsync(new UserDefinedIndexSpec + { + Name = $"s6-idx-{i:D3}", + EventTypes = new HashSet { "UdiA", "UdiB" } + }); + } + catch (Exception ex) { failures.Add($"#{i}: {ex.GetType().Name} {ex.Message}"); } + } + sw.Stop(); + output.WriteLine($"SPIKE-6 created {count - failures.Count}/{count} indexes in {sw.ElapsedMilliseconds} ms; " + + $"failures={failures.Count}"); + foreach (var f in failures.Take(5)) output.WriteLine(" " + f); + + // Sample the FIRST and LAST index: both must backfill to the full 100 and read in commit order — proving + // no degradation/starvation under many concurrent index backfills. + foreach (var idx in new[] { "s6-idx-000", $"s6-idx-{count - 1:D3}" }) + { + await src.WaitUntilReadyAsync(idx, expectedCount: 100, TimeSpan.FromSeconds(120)); + var ords = await ReadOrdsAsync(src, idx); + output.WriteLine($"SPIKE-6 sample {idx}: count={ords.Count}, ordered={ords.SequenceEqual(Enumerable.Range(0, 100))}"); + ords.Should().Equal(Enumerable.Range(0, 100), $"{idx} must read the full merge in commit order"); + } + + failures.Should().BeEmpty("creating ~60 indexes on one node must not be rejected"); + } + + // ===================================================================================================== + // SPIKE-7 (team-lead's "SPIKE-5 persistent" / arch's persistent-sub open item) — THE REAL BLOCKER for + // shape-A merges that go through SubscribeEventHandlerPersistently (ProcessManager Inbox/Outbox, competing + // consumers, server checkpoints). EMPIRICAL RESULT: a SERVER-CHECKPOINTED PERSISTENT subscription over + // filtered $all (StreamFilter.Prefix("$idx-user-") + resolveLinkTos) is CREATABLE and connects with + // NO error — but delivers ZERO index entries (catch-up AND live). Same silent-empty as SubscribeToStream on + // the index stream (SPIKE-2a). The Spike7_controls test proves this is NOT a broken harness (a persistent + // filtered-$all sub over a NON-link filter delivers fine) — persistent subscriptions simply do not see the + // $idx-user-* index links, by ANY mechanism (filtered-$all resolve/no-resolve, or direct CreateToStream). + // This test ENCODES that measured reality: create succeeds, subscribe yields nothing. Contrast SPIKE-2b + // (catch-up SubscribeToAll+filter) which DOES deliver — so index-backing is a CATCH-UP-only capability. + // ===================================================================================================== + [Fact] + public async Task Spike7_persistent_subscription_over_filtered_all_delivers_nothing_blocker() + { + await using var stores = new Store(); + var (client, ps, conn) = await stores.NewWithPsAsync("s7"); + + await client.AppendToStreamAsync("S7-a", StreamState.NoStream, + [Typed("UdiA", 0), Typed("UdiB", 1), Typed("UdiA", 2)]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "s7-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 3, TimeSpan.FromSeconds(30)); + + var indexStream = UserDefinedIndexSource.IndexStream(spec.Name); + const string group = "rm-group"; + + // The persistent group over FILTERED $all with resolveLinkTos IS creatable (this is the trap — it looks + // like it should work). + string? createError = null; + try + { + await ps.CreateToAllAsync(group, StreamFilter.Prefix(indexStream), + new PersistentSubscriptionSettings(resolveLinkTos: true, startFrom: Position.Start, + checkPointLowerBound: 1, checkPointAfter: TimeSpan.FromMilliseconds(100))); + } + catch (Exception ex) { createError = $"{ex.GetType().Name}: {ex.Message}"; } + output.WriteLine($"SPIKE-7 CreateToAllAsync(filtered) error: {createError ?? ""}"); + createError.Should().BeNull("the persistent group IS creatable — the failure is delivery, not creation"); + + var received = new ConcurrentQueue(); + string? pumpError = null; + using var cts = new CancellationTokenSource(); + var pump = Task.Run(async () => + { + try + { + await using var sub = ps.SubscribeToAll(group, cancellationToken: cts.Token); + await foreach (var e in sub.WithCancellation(cts.Token)) + { + received.Enqueue(JsonNode.Parse(e.Event.Data.Span)?["Ord"]?.GetValue() ?? -1); + await sub.Ack(e); + } + } + catch (OperationCanceledException) { } + catch (Exception ex) { pumpError = $"{ex.GetType().Name}: {ex.Message}"; } + }); + + // Wait for catch-up that will NOT come; also append live to prove live push is dead too. + var caughtUp = await WaitUntil(() => Task.FromResult(received.Count >= 3), TimeSpan.FromSeconds(15)); + await client.AppendToStreamAsync("S7-b", StreamState.NoStream, [Typed("UdiB", 3), Typed("UdiA", 4)]); + var gotLive = await WaitUntil(() => Task.FromResult(received.Count >= 5), TimeSpan.FromSeconds(15)); + + cts.Cancel(); + await pump; + + output.WriteLine($"SPIKE-7 pump error: {pumpError ?? ""}"); + output.WriteLine($"SPIKE-7 persistent delivery — caughtUp={caughtUp} gotLive={gotLive} received=[{string.Join(",", received)}]"); + pumpError.Should().BeNull("subscribing does not fault — it just silently delivers nothing"); + received.Should().BeEmpty( + "MEASURED BLOCKER: a persistent subscription over filtered $all delivers ZERO $idx-user-* index entries " + + "(catch-up and live) — no error, no events. Persistent (server-checkpointed) read models CANNOT be " + + "index-backed; only catch-up (client-checkpointed) subscriptions see the index (SPIKE-2b). See " + + "Spike7_controls for the harness-sanity control (a NON-link filter delivers fine)."); + } + + // SPIKE-7 CONTROLS — distinguish "my persistent-sub harness is broken" from "persistent-sub-to-$idx-user + // genuinely delivers nothing". Counts what a filtered persistent $all subscription delivers for: + // A) a NORMAL event-type filter (domain events directly in $all) -> proves the harness works at all + // B) StreamFilter.Prefix($idx-user-*) with resolveLinkTos:FALSE -> are the raw index LINK events deliverable + // C) StreamFilter.Prefix($idx-user-*) with resolveLinkTos:TRUE -> the failing case, re-measured alongside + private async Task DrainPersistentAsync(KurrentDBPersistentSubscriptionsClient ps, string group, + int expect, TimeSpan timeout) + { + var got = new ConcurrentQueue(); + using var cts = new CancellationTokenSource(); + var pump = Task.Run(async () => + { + try + { + await using var sub = ps.SubscribeToAll(group, cancellationToken: cts.Token); + await foreach (var e in sub.WithCancellation(cts.Token)) + { + got.Enqueue(1); + await sub.Ack(e); + } + } + catch (OperationCanceledException) { } + }); + await WaitUntil(() => Task.FromResult(got.Count >= expect), timeout); + await Task.Delay(300); + cts.Cancel(); + await pump; + return got.Count; + } + + [Fact] + public async Task Spike7_controls_isolate_whether_persistent_filtered_all_harness_works() + { + await using var stores = new Store(); + var (client, ps, conn) = await stores.NewWithPsAsync("s7c"); + + await client.AppendToStreamAsync("S7c-a", StreamState.NoStream, + [Typed("UdiA", 0), Typed("UdiB", 1), Typed("UdiA", 2)]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "s7c-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 3, TimeSpan.FromSeconds(30)); + var indexStream = UserDefinedIndexSource.IndexStream(spec.Name); + + var settings = () => new PersistentSubscriptionSettings(resolveLinkTos: false, startFrom: Position.Start, + checkPointLowerBound: 1, checkPointAfter: TimeSpan.FromMilliseconds(100)); + + // A) Normal event-type filter over $all (domain events themselves). Proves the harness delivers. + await ps.CreateToAllAsync("ctrl-a", EventTypeFilter.Prefix("Udi"), settings()); + var a = await DrainPersistentAsync(ps, "ctrl-a", 3, TimeSpan.FromSeconds(20)); + output.WriteLine($"SPIKE-7 CONTROL A (EventTypeFilter 'Udi', domain events): delivered {a} (expect 3)"); + + // B) $idx-user prefix filter, resolveLinkTos FALSE — raw link events. + await ps.CreateToAllAsync("ctrl-b", StreamFilter.Prefix(indexStream), settings()); + var b = await DrainPersistentAsync(ps, "ctrl-b", 3, TimeSpan.FromSeconds(20)); + output.WriteLine($"SPIKE-7 CONTROL B ($idx-user prefix, resolveLinkTos=FALSE): delivered {b} (expect 3 if links deliverable)"); + + // C) $idx-user prefix filter, resolveLinkTos TRUE — the failing case, measured here too. + await ps.CreateToAllAsync("ctrl-c", StreamFilter.Prefix(indexStream), + new PersistentSubscriptionSettings(resolveLinkTos: true, startFrom: Position.Start, + checkPointLowerBound: 1, checkPointAfter: TimeSpan.FromMilliseconds(100))); + var c = await DrainPersistentAsync(ps, "ctrl-c", 3, TimeSpan.FromSeconds(20)); + output.WriteLine($"SPIKE-7 CONTROL C ($idx-user prefix, resolveLinkTos=TRUE): delivered {c} (expect 3)"); + + // D) PERSISTENT subscription created DIRECTLY on the $idx-user stream (CreateToStreamAsync) — the + // persistent analog of SPIKE-2a's SubscribeToStream, measured rather than inferred. + var d = -1; string? dErr = null; + try + { + await ps.CreateToStreamAsync(indexStream, "ctrl-d", + new PersistentSubscriptionSettings(resolveLinkTos: true, startFrom: StreamPosition.Start, + checkPointLowerBound: 1, checkPointAfter: TimeSpan.FromMilliseconds(100))); + var got = new ConcurrentQueue(); + using var cts = new CancellationTokenSource(); + var pump = Task.Run(async () => + { + try + { + await using var sub = ps.SubscribeToStream(indexStream, "ctrl-d", cancellationToken: cts.Token); + await foreach (var e in sub.WithCancellation(cts.Token)) { got.Enqueue(1); await sub.Ack(e); } + } + catch (OperationCanceledException) { } + }); + await WaitUntil(() => Task.FromResult(got.Count >= 3), TimeSpan.FromSeconds(20)); + await Task.Delay(300); cts.Cancel(); await pump; + d = got.Count; + } + catch (Exception ex) { dErr = $"{ex.GetType().Name}: {ex.Message}"; } + output.WriteLine($"SPIKE-7 CONTROL D (persistent CreateToStream on $idx-user): delivered {d} (expect 3), err={dErr ?? ""}"); + + output.WriteLine($"SPIKE-7 CONTROLS SUMMARY: A(domain)={a} B(link,noresolve)={b} C(link,resolve)={c} D(persist-to-stream)={d}"); + // A is the harness sanity check — it MUST deliver. B/C/D are the actual findings (logged, not asserted hard). + a.Should().Be(3, "the persistent filtered-$all harness itself works when the filter matches non-link events"); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/MicroPlumberd.Migration.Tests.csproj b/src/MicroPlumberd.Migration.Tests/MicroPlumberd.Migration.Tests.csproj new file mode 100644 index 0000000..a007cb1 --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/MicroPlumberd.Migration.Tests.csproj @@ -0,0 +1,35 @@ + + + + net10.0 + enable + enable + false + true + + + + + + + runtime; build; native; contentfiles; analyzers; buildtransitive + all + + + + runtime; build; native; contentfiles; analyzers; buildtransitive + all + + + + + + + + + + + + + diff --git a/src/MicroPlumberd.Migration.Tests/MigrationCopyIntegrationTests.cs b/src/MicroPlumberd.Migration.Tests/MigrationCopyIntegrationTests.cs new file mode 100644 index 0000000..4c655f3 --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/MigrationCopyIntegrationTests.cs @@ -0,0 +1,471 @@ +using System.Text; +using System.Text.Json; +using System.Text.Json.Nodes; +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Testing; +using Xunit; + +namespace MicroPlumberd.Migration.Tests; + +/// +/// End-to-end copy-engine tests against real KurrentDB containers (via ). +/// Each test provisions its own isolated SOURCE and DEST stores because the copy is whole-store. +/// +[Trait("Category", "Integration")] +public class MigrationCopyIntegrationTests +{ + private sealed class Stores : IAsyncDisposable + { + private readonly List _servers = new(); + + public async Task NewStoreAsync(string tag) + { + var srv = EventStoreServer.Create($"mp-mig-{tag}-{Guid.NewGuid():N}"); + _servers.Add(srv); + await srv.StartInDocker(inMemory: true); + return new KurrentDBClient(srv.GetEventStoreSettings()); + } + + public async ValueTask DisposeAsync() + { + foreach (var s in _servers) await s.DisposeAsync(); + } + } + + private static Task AppendAsync(KurrentDBClient c, string stream, StreamState expected, + string type, string dataJson, string? metaJson = null) + { + var ev = new EventData(Uuid.NewUuid(), type, Encoding.UTF8.GetBytes(dataJson), + Encoding.UTF8.GetBytes(metaJson ?? "{}")); + return c.AppendToStreamAsync(stream, expected, [ev]); + } + + private static async Task> ReadStreamAsync(KurrentDBClient c, string stream) + { + var res = c.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: false); + if (await res.ReadState == ReadState.StreamNotFound) return new List(); + var list = new List(); + await foreach (var e in res) list.Add(e); + return list; + } + + private sealed class InlineMigration(string id, Action build) : Migration + { + public override string Id => id; + public override void Migrate(IMigrationBuilder b) => build(b); + } + + [Fact] + public async Task Copy_preserves_order_and_version_stamps_MigratedAt_and_applies_rules() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("src"); + var dst = await stores.NewStoreAsync("dst"); + + // Order-1: three events — the middle one carries a correlation id we expect to survive verbatim. + await AppendAsync(src, "Order-1", StreamState.NoStream, "OrderPlaced", "{\"n\":1}"); + await AppendAsync(src, "Order-1", 0ul, "OrderPlaced", + "{\"n\":2}", "{\"$correlationId\":\"11111111-1111-1111-1111-111111111111\"}"); + await AppendAsync(src, "Order-1", 1ul, "OldName", "{\"keep\":true}"); + // Widget-1: transformed. Junk-1: dropped whole-stream. + await AppendAsync(src, "Widget-1", StreamState.NoStream, "Widget", "{\"a\":1}"); + await AppendAsync(src, "Junk-1", StreamState.NoStream, "Whatever", "{}"); + + var migration = new InlineMigration("0001_mixed", b => + { + b.DropStream("Junk-1"); + b.RenameType("OldName", "NewName"); + b.TransformJson("Widget", (data, meta) => + { + data["migrated"] = true; + meta["note"] = "widget-touched"; + }); + }); + + var result = await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: false); + + // Verification report is clean and drops are accounted for. + result.Verification!.AllOk.Should().BeTrue(result.Verification.Format()); + result.Copy.Dropped.Should().Be(1); + result.Copy.Kept.Should().Be(4); + + // Order-1 preserved: 3 events, contiguous versions 0,1,2, order intact, type renamed on the 3rd. + var order = await ReadStreamAsync(dst, "Order-1"); + order.Should().HaveCount(3); + order.Select(e => e.Event.EventNumber.ToUInt64()).Should().Equal(0ul, 1ul, 2ul); + order[2].Event.EventType.Should().Be("NewName"); + + // MigratedAt stamped on every copied event; original correlation id preserved verbatim. + foreach (var e in order) + { + var meta = JsonNode.Parse(e.Event.Metadata.Span)!.AsObject(); + meta.ContainsKey("MigratedAt").Should().BeTrue(); + } + var midMeta = JsonNode.Parse(order[1].Event.Metadata.Span)!.AsObject(); + midMeta["$correlationId"]!.GetValue().Should().Be("11111111-1111-1111-1111-111111111111"); + + // Transform applied to raw JSON (payload + metadata). + var widget = await ReadStreamAsync(dst, "Widget-1"); + var wData = JsonNode.Parse(widget[0].Event.Data.Span)!.AsObject(); + wData["migrated"]!.GetValue().Should().BeTrue(); + var wMeta = JsonNode.Parse(widget[0].Event.Metadata.Span)!.AsObject(); + wMeta["note"]!.GetValue().Should().Be("widget-touched"); + + // Dropped stream absent at destination. + (await ReadStreamAsync(dst, "Junk-1")).Should().BeEmpty(); + + // History recorded. + result.NewlyApplied.Should().ContainSingle(r => r.Id == "0001_mixed"); + result.NewlyApplied[0].Dropped.Should().Be(1); + } + + [Fact] + public async Task No_rule_event_is_copied_byte_verbatim_and_MigratedAt_is_still_stamped() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("vsrc"); + var dst = await stores.NewStoreAsync("vdst"); + + // Deliberately NON-canonical JSON (extra whitespace) so a re-serialise would change the bytes — that + // makes byte-equality at the destination a real proof the payload was NOT parsed/re-serialised. + const string spacedData = "{ \"a\" : 1 , \"b\" : \"x\" }"; + const string meta = "{\"$correlationId\":\"11111111-1111-1111-1111-111111111111\"}"; + await AppendAsync(src, "Keep-1", StreamState.NoStream, "Kept", spacedData, meta); + + // DropStreams-only migration, targeting a DIFFERENT stream → nothing about Keep-1 is parsed. + var migration = new InlineMigration("0001_drop_other", b => b.DropStream("Junk-1")); + await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: false); + + var copied = await ReadStreamAsync(dst, "Keep-1"); + copied.Should().HaveCount(1); + var e = copied[0].Event; + + // DATA copied byte-for-byte — proves the common path neither parses nor re-serialises the payload. + e.Data.ToArray().Should().Equal(Encoding.UTF8.GetBytes(spacedData)); + + // METADATA still carries MigratedAt (stamped without a JsonNode round-trip) and preserves the original + // correlation id verbatim. + var m = JsonNode.Parse(e.Metadata.Span)!.AsObject(); + m.ContainsKey("MigratedAt").Should().BeTrue(); + m["$correlationId"]!.GetValue().Should().Be("11111111-1111-1111-1111-111111111111"); + } + + [Fact] + public async Task Tombstoned_source_stream_is_handled_without_crashing() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("tsrc"); + var dst = await stores.NewStoreAsync("tdst"); + + await AppendAsync(src, "Keep-1", StreamState.NoStream, "Kept", "{\"ok\":1}"); + await AppendAsync(src, "Offer-of-dead", StreamState.NoStream, "OfferCreated", "{\"x\":1}"); + await AppendAsync(src, "Offer-of-dead", 0ul, "OfferChanged", "{\"x\":2}"); + // Hard-delete (tombstone) the offer stream — this is the real :5081 condition. + await src.TombstoneAsync("Offer-of-dead", StreamState.Any); + + var migration = new InlineMigration("0001_drop_dead", b => b.DropStreams("Offer-of-dead")); + + // Must NOT throw despite the tombstoned stream / deleted-stream artifacts in $all. + var result = await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: false); + + (await ReadStreamAsync(dst, "Keep-1")).Should().HaveCount(1); + (await ReadStreamAsync(dst, "Offer-of-dead")).Should().BeEmpty(); + result.Verification!.AllOk.Should().BeTrue(result.Verification.Format()); + } + + [Fact] + public async Task DryRun_writes_nothing_but_reports_counts() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("dsrc"); + var dst = await stores.NewStoreAsync("ddst"); + + await AppendAsync(src, "Keep-1", StreamState.NoStream, "Kept", "{}"); + await AppendAsync(src, "Junk-1", StreamState.NoStream, "Whatever", "{}"); + + var migration = new InlineMigration("0001_drop_junk", b => b.DropStream("Junk-1")); + + var result = await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: true); + + result.DryRun.Should().BeTrue(); + result.Copy.Kept.Should().Be(1); + result.Copy.Dropped.Should().Be(1); + result.Copy.MigrationStats["0001_drop_junk"].Dropped.Should().Be(1); + result.Verification.Should().BeNull(); + result.NewlyApplied.Should().BeEmpty(); + + // Destination truly untouched. + (await ReadStreamAsync(dst, "Keep-1")).Should().BeEmpty(); + var all = dst.ReadAllAsync(Direction.Forwards, Position.Start); + var userEvents = 0; + await foreach (var e in all) + if (e.Event is { } er && er.EventStreamId.Length > 0 && er.EventStreamId[0] != '$') userEvents++; + userEvents.Should().Be(0); + } + + // NOTE: the [OutputStream] merge/read-model replay proof lives in IncidentReplayIntegrationTests, which + // rides the ACTUAL ERP path — the $et JOIN projection (TryCreateJoinProjection → fromStreams(['$et-…']) + // .linkTo(X)) — not the $ce split projection. A $ce-based test was removed as it exercised a path the + // app does not use. + + [Fact] + public async Task History_travels_with_store_enabling_pending_detection_and_checksum_guard() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("hsrc"); + var d1 = await stores.NewStoreAsync("hd1"); + + await AppendAsync(src, "Keep-1", StreamState.NoStream, "Kept", "{}"); + await AppendAsync(src, "Junk-1", StreamState.NoStream, "Whatever", "{}"); + + var m1 = new InlineMigration("0001_drop_junk", b => b.DropStream("Junk-1")); + + // First run applies m1 → d1 gets history. + var first = await new MigrationRunner().RunAsync(src, d1, [m1], dryRun: false); + first.NewlyApplied.Should().ContainSingle(r => r.Id == "0001_drop_junk"); + + // Re-migrating d1 with the SAME migration set → nothing pending, and history carries forward. + var d2 = await stores.NewStoreAsync("hd2"); + var second = await new MigrationRunner().RunAsync(d1, d2, [m1], dryRun: false); + second.PendingMigrationIds.Should().BeEmpty(); + second.NewlyApplied.Should().BeEmpty(); + var historyOnD2 = await ReadStreamAsync(d2, MigrationHistory.StreamName); + historyOnD2.Should().ContainSingle(e => e.Event.EventType == nameof(MigrationApplied)); + + // Editing an already-applied migration (same Id, different rules) is refused. + var d3 = await stores.NewStoreAsync("hd3"); + var edited = new InlineMigration("0001_drop_junk", b => b.DropStream("Something-Else")); + var act = async () => await new MigrationRunner().RunAsync(d1, d3, [edited], dryRun: false); + (await act.Should().ThrowAsync()) + .Which.MigrationId.Should().Be("0001_drop_junk"); + } + + // M1: a large CONTIGUOUS single-stream run must flush MID-STREAM (batch caps) yet still land every event, + // in order, with GAPLESS versions — proving multiple appends to one stream continue the expected version. + [Fact] + public async Task Large_single_stream_run_flushes_mid_stream_and_preserves_gapless_order() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("bsrc"); + var dst = await stores.NewStoreAsync("bdst"); + + const int n = 1200; // > MaxBatchCount (1000) → forces at least one mid-stream flush + var expected = StreamState.NoStream; + for (var offset = 0; offset < n; offset += 200) + { + var chunk = Enumerable.Range(offset, Math.Min(200, n - offset)) + .Select(i => new EventData(Uuid.NewUuid(), "Bulk", + Encoding.UTF8.GetBytes($"{{\"i\":{i}}}"), Encoding.UTF8.GetBytes("{}"))) + .ToArray(); + await src.AppendToStreamAsync("Bulk-1", expected, chunk); + expected = (ulong)(offset + chunk.Length - 1); + } + + var result = await new MigrationRunner().RunAsync(src, dst, Array.Empty(), dryRun: false); + result.Copy.Kept.Should().Be(n); + result.Verification!.AllOk.Should().BeTrue(result.Verification.Format()); + // A single stream of n>MaxBatchCount events MUST have been split across multiple appends (a single big + // append is impossible under the count cap) — gaplessness alone would be consistent with one append. + result.Copy.DestAppendCount.Should().BeGreaterThan(1, "the count cap must split one stream into >1 append"); + + var copied = await ReadStreamAsync(dst, "Bulk-1"); + copied.Should().HaveCount(n); + // Gapless versions 0..n-1 AND payload order preserved across the mid-stream flushes. + copied.Select(e => e.Event.EventNumber.ToUInt64()).Should().Equal(Enumerable.Range(0, n).Select(i => (ulong)i)); + copied.Select(e => JsonNode.Parse(e.Event.Data.Span)!["i"]!.GetValue()) + .Should().Equal(Enumerable.Range(0, n)); + } + + // M2: an unparseable-but-declared-JSON event, in a store where a DropEvent rule forces parse-all, must be + // copied VERBATIM (byte-for-byte), NOT dropped. Only the rule's explicit target is dropped. + [Fact] + public async Task Unparseable_json_event_is_copied_verbatim_not_dropped_even_with_a_DropEvent_rule() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("usrc"); + var dst = await stores.NewStoreAsync("udst"); + + var invalidBytes = Encoding.UTF8.GetBytes("{ this is NOT valid json "); + await src.AppendToStreamAsync("S-1", StreamState.NoStream, + [new EventData(Uuid.NewUuid(), "Weird", invalidBytes, Encoding.UTF8.GetBytes("{}"))]); // json content-type + await AppendAsync(src, "S-1", 0ul, "Good", "{\"ok\":1}"); + await AppendAsync(src, "S-1", 1ul, "Noise", "{}"); // this one IS dropped by the rule + + // A DropEvent rule → forces parse-all; it drops only type "Noise". + var migration = new InlineMigration("0001_drop_noise", b => b.DropEvent(e => e.Type == "Noise")); + var result = await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: false); + + result.Copy.UnparseableVerbatim.Should().Be(1, "the unparseable event is counted (as verbatim, not lost)"); + result.Copy.Dropped.Should().Be(1, "only the Noise event is dropped"); + result.Verification!.AllOk.Should().BeTrue(result.Verification.Format()); + + var copied = await ReadStreamAsync(dst, "S-1"); + copied.Select(e => e.Event.EventType).Should().Equal("Weird", "Good"); // Noise dropped; Weird kept + // The unparseable payload is preserved byte-for-byte (proves verbatim copy, no corruption). + copied[0].Event.Data.ToArray().Should().Equal(invalidBytes); + } + + // V1 merge: two source streams RENAMED into ONE dest stream keep their interleaved $all/arrival order with + // gapless version on the merged stream. + [Fact] + public async Task RenameStream_merges_two_sources_into_one_preserving_interleaved_order_and_gapless_version() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("msrc"); + var dst = await stores.NewStoreAsync("mdst"); + + // Interleaved arrival across A-1 and B-1: A0, B0, A1, B1. + await AppendAsync(src, "A-1", StreamState.NoStream, "T", "{\"v\":\"a0\"}"); + await AppendAsync(src, "B-1", StreamState.NoStream, "T", "{\"v\":\"b0\"}"); + await AppendAsync(src, "A-1", 0ul, "T", "{\"v\":\"a1\"}"); + await AppendAsync(src, "B-1", 0ul, "T", "{\"v\":\"b1\"}"); + + var migration = new InlineMigration("0001_merge", b => + { + b.RenameStream("A-1", "D-1"); + b.RenameStream("B-1", "D-1"); + }); + var result = await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: false); + result.Verification!.AllOk.Should().BeTrue(result.Verification.Format()); + + var merged = await ReadStreamAsync(dst, "D-1"); + merged.Should().HaveCount(4); + merged.Select(e => e.Event.EventNumber.ToUInt64()).Should().Equal(0ul, 1ul, 2ul, 3ul); // gapless + merged.Select(e => JsonNode.Parse(e.Event.Data.Span)!["v"]!.GetValue()) + .Should().Equal("a0", "b0", "a1", "b1"); // original $all arrival order + (await ReadStreamAsync(dst, "A-1")).Should().BeEmpty(); + (await ReadStreamAsync(dst, "B-1")).Should().BeEmpty(); + } + + // M4: the write-fidelity checksum catches a dest whose CONTENT differs from the copy intent even when the + // COUNT matches (a count-only verifier would pass it). Copy once cleanly; copy again with a transform that + // alters the bytes; verify the altered dest against the CLEAN copy's hashes → MISMATCH on the changed + // streams, while the clean dest verifies OK. + [Fact] + public async Task Verification_write_fidelity_checksum_catches_a_content_miscopy_that_counts_miss() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("csrc"); + var clean = await stores.NewStoreAsync("cclean"); + var altered = await stores.NewStoreAsync("calt"); + + await AppendAsync(src, "S-1", StreamState.NoStream, "T", "{\"n\":1}"); + await AppendAsync(src, "S-1", 0ul, "T", "{\"n\":2}"); + await AppendAsync(src, "S-2", StreamState.NoStream, "U", "{\"n\":9}"); + + var reserved = new HashSet(StringComparer.Ordinal) { MigrationHistory.StreamName }; + + // Clean copy — its DestStreamHashes capture the faithful write intent. + var cleanResult = await new MigrationRunner().RunAsync(src, clean, Array.Empty(), dryRun: false); + + // Positive control: the clean dest verifies OK against its own hashes. + var okReport = await new Verifier().VerifyAsync(clean, cleanResult.Copy, reserved); + okReport.AllOk.Should().BeTrue(okReport.Format()); + okReport.Streams.Should().OnlyContain(s => s.HashOk); + + // A second copy that ALTERS type-T bytes (adds a field) — same streams, same counts, different content. + var alterMigration = new InlineMigration("0001_alter", b => b.TransformJson("T", (d, _) => d["x"] = 1)); + await new MigrationRunner().RunAsync(src, altered, [alterMigration], dryRun: false); + + // Verify the ALTERED dest against the CLEAN copy's hashes: counts match, but the checksum flags S-1. + var badReport = await new Verifier().VerifyAsync(altered, cleanResult.Copy, reserved); + badReport.AllOk.Should().BeFalse("the checksum must catch the altered content the count check misses"); + var s1 = badReport.Streams.Single(s => s.DestStream == "S-1"); + s1.ExpectedCount.Should().Be(s1.ActualCount); // count is fine … + s1.HashOk.Should().BeFalse(); // … but the write-fidelity checksum is not + s1.Status.Should().Be(VerificationStatus.Mismatch); + // Precision: S-2 (type U) was NOT altered by the T-transform, so its checksum still matches. + badReport.Streams.Single(s => s.DestStream == "S-2").HashOk.Should().BeTrue(); + } + + // M1 byte cap: LARGE payloads must split a single stream on the BYTES branch (not the count branch), and a + // single event LARGER than the whole cap must be appended ALONE. Uses one stream so DestAppendCount is + // unambiguous: [1.5 MiB, 0.6 MiB, 0.6 MiB] → the oversize event flushes alone, then each 0.6 MiB event + // exceeds the 1 MiB cap when added to a non-empty batch → 3 separate appends. + [Fact] + public async Task Byte_cap_splits_large_payloads_and_appends_an_oversize_event_alone() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("ysrc"); + var dst = await stores.NewStoreAsync("ydst"); + + static byte[] Blob(int bytes) => + Encoding.UTF8.GetBytes("{\"blob\":\"" + new string('x', bytes) + "\"}"); + + var big = Blob(1_200 * 1024); // > 1 MiB cap → must be appended alone (kept well under single-event limits) + var mid1 = Blob(600 * 1024); + var mid2 = Blob(600 * 1024); + await src.AppendToStreamAsync("Byte-1", StreamState.NoStream, + [ + new EventData(Uuid.NewUuid(), "T", big, Encoding.UTF8.GetBytes("{}")), + new EventData(Uuid.NewUuid(), "T", mid1, Encoding.UTF8.GetBytes("{}")), + new EventData(Uuid.NewUuid(), "T", mid2, Encoding.UTF8.GetBytes("{}")), + ]); + + var result = await new MigrationRunner().RunAsync(src, dst, Array.Empty(), dryRun: false); + result.Verification!.AllOk.Should().BeTrue(result.Verification.Format()); + // BYTES branch (not count: only 3 events, far below MaxBatchCount) split this one stream into 3 appends, + // the first carrying the oversize event alone. + result.Copy.DestAppendCount.Should().Be(3, "the byte cap must split large payloads (incl. an oversize event alone)"); + + var copied = await ReadStreamAsync(dst, "Byte-1"); + copied.Should().HaveCount(3); + copied.Select(e => e.Event.EventNumber.ToUInt64()).Should().Equal(0ul, 1ul, 2ul); // gapless across splits + copied[0].Event.Data.ToArray().Should().Equal(big); // oversize event copied intact, in order + copied[1].Event.Data.ToArray().Should().Equal(mid1); + copied[2].Event.Data.ToArray().Should().Equal(mid2); + } + + // M4 direct reorder: two dest events with IDENTICAL bytes but SWAPPED order must fail the write-fidelity + // checksum (the hash is order-sensitive) even though the count is unchanged — the exact property the + // content-alteration test does not isolate. + [Fact] + public async Task Verification_checksum_catches_reordered_dest_events() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("rsrc"); + var clean = await stores.NewStoreAsync("rclean"); + var swapped = await stores.NewStoreAsync("rswap"); + + await AppendAsync(src, "S-1", StreamState.NoStream, "T", "{\"v\":1}"); + await AppendAsync(src, "S-1", 0ul, "T", "{\"v\":2}"); + + var reserved = new HashSet(StringComparer.Ordinal) { MigrationHistory.StreamName }; + var cleanResult = await new MigrationRunner().RunAsync(src, clean, Array.Empty(), dryRun: false); + (await new Verifier().VerifyAsync(clean, cleanResult.Copy, reserved)).AllOk.Should().BeTrue(); + + // Hand-build a dest with the SAME two (EventType, Data) pairs but in SWAPPED order (v2 then v1). + await swapped.AppendToStreamAsync("S-1", StreamState.NoStream, + [ + new EventData(Uuid.NewUuid(), "T", Encoding.UTF8.GetBytes("{\"v\":2}"), Encoding.UTF8.GetBytes("{}")), + new EventData(Uuid.NewUuid(), "T", Encoding.UTF8.GetBytes("{\"v\":1}"), Encoding.UTF8.GetBytes("{}")), + ]); + + var report = await new Verifier().VerifyAsync(swapped, cleanResult.Copy, reserved); + var s1 = report.Streams.Single(s => s.DestStream == "S-1"); + s1.ExpectedCount.Should().Be(s1.ActualCount); // count is identical (2 == 2) … + s1.HashOk.Should().BeFalse(); // … but the order-sensitive checksum catches the swap + report.AllOk.Should().BeFalse(); + } + + // m8a: a RenameStream/RenameType into the $-system namespace (events would vanish as system) is rejected. + [Fact] + public async Task Rename_into_system_namespace_is_rejected() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("gsrc"); + var dst = await stores.NewStoreAsync("gdst"); + await AppendAsync(src, "A-1", StreamState.NoStream, "T", "{}"); + + var badStream = new InlineMigration("0001_bad_stream", b => b.RenameStream("A-1", "$sneaky")); + var badType = new InlineMigration("0001_bad_type", b => b.RenameType("T", "$sneaky")); + + var s = async () => await new MigrationRunner().RunAsync(src, dst, [badStream], dryRun: true); + await s.Should().ThrowAsync(); + + var t = async () => await new MigrationRunner().RunAsync(src, dst, [badType], dryRun: true); + await t.Should().ThrowAsync(); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/MigrationProgressTests.cs b/src/MicroPlumberd.Migration.Tests/MigrationProgressTests.cs new file mode 100644 index 0000000..bef6f64 --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/MigrationProgressTests.cs @@ -0,0 +1,103 @@ +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Testing; +using MicroPlumberd.Tests.App.Domain; +using Xunit; +using Xunit.Abstractions; + +namespace MicroPlumberd.Migration.Tests; + +/// +/// Integration test for the live status + progress API ( exposed via +/// ). Runs a real migration over real KurrentDB and asserts the status +/// snapshot transitions Pending → Running → Completed, that per-migration items transition +/// Pending → Running → Completed, and that per-phase progress advances and reaches 100%. +/// +[Trait("Category", "Integration")] +public class MigrationProgressTests(ITestOutputHelper output) +{ + private sealed class Server : IAsyncDisposable + { + public required EventStoreServer Es { get; init; } + public required KurrentDBClient Client { get; init; } + public required IPlumber Plumber { get; init; } + + public static async Task StartAsync(string tag) + { + var es = EventStoreServer.Create($"mp-prog-{tag}-{Guid.NewGuid():N}"); + await es.StartInDocker(inMemory: true); + var s = es.GetEventStoreSettings(); + return new Server + { + Es = es, + Client = new KurrentDBClient(s), + Plumber = global::MicroPlumberd.Plumber.Create(s) + }; + } + + public async ValueTask DisposeAsync() => await Es.DisposeAsync(); + } + + /// A pending migration that drops one unused stream — enough to exercise item state transitions. + private sealed class NoopDrop : Migration + { + public override string Id => "0001_progress_probe"; + public override void Migrate(IMigrationBuilder b) => b.DropStream("Nonexistent-stream-xyz"); + } + + [Fact] + public async Task Status_snapshot_transitions_pending_running_completed_and_progress_reaches_100() + { + await using var src = await Server.StartAsync("src"); + await using var dst = await Server.StartAsync("dst"); + + // Seed a batch of aggregate events so the event-copy phase has real work + a non-trivial position. + const int n = 40; + for (var i = 0; i < n; i++) + await src.Plumber.SaveNew(FooAggregate.Open($"foo-{i}", Guid.NewGuid())); + + var runner = new MigrationRunner(); + + // BEFORE the run: the snapshot is Pending / NotStarted. + runner.CurrentStatus.RunState.Should().Be(MigrationRunState.Pending); + runner.CurrentStatus.Phase.Should().Be(MigrationPhase.NotStarted); + + // Capture every pushed snapshot (published synchronously on the running task; guard with a lock anyway). + var snapshots = new List(); + var gate = new object(); + runner.Progress.ProgressChanged += (_, s) => { lock (gate) snapshots.Add(s); }; + + // SIMPLE migration (no projection pre-copy — the app regenerates merge streams on boot via fromAll). + var result = await runner.RunAsync(src.Client, dst.Client, [new NoopDrop()], dryRun: false); + + List seen; + lock (gate) seen = snapshots.ToList(); + output.WriteLine("RunState sequence: " + string.Join(" → ", seen.Select(s => s.RunState).Distinct())); + output.WriteLine($"final EventCopy: copied={seen[^1].EventCopy?.EventsCopied} head-based %={seen[^1].EventCopy?.PercentComplete:0.0}"); + + // (1) Overall run state: Pending (before) → Running (during) → Completed (after). + runner.CurrentStatus.RunState.Should().Be(MigrationRunState.Completed); + runner.CurrentStatus.Phase.Should().Be(MigrationPhase.Completed); + runner.CurrentStatus.FinishedUtc.Should().NotBeNull(); + seen.Should().Contain(s => s.RunState == MigrationRunState.Running, "the run must be observed Running mid-flight"); + seen[^1].RunState.Should().Be(MigrationRunState.Completed, "the last pushed snapshot is Completed"); + + // (2) The pending migration item transitions Pending → Running → Completed. + seen.Should().Contain(s => s.Migrations.Any(m => m.Id == "0001_progress_probe" && m.State == MigrationItemState.Running), + "the migration must be observed Running"); + var finalItem = runner.CurrentStatus.Migrations.Single(m => m.Id == "0001_progress_probe"); + finalItem.State.Should().Be(MigrationItemState.Completed); + finalItem.StartedUtc.Should().NotBeNull(); + finalItem.FinishedUtc.Should().NotBeNull(); + + // (3) Event-copy progress advanced from 0 and reached ~100% (position-based) with the full written count. + var ecStates = seen.Where(s => s.EventCopy is not null).Select(s => s.EventCopy!).ToList(); + ecStates.Should().NotBeEmpty(); + ecStates.Max(e => e.EventsCopied).Should().BeGreaterThan(0, "events were copied"); + var finalEc = runner.CurrentStatus.EventCopy!; + finalEc.EventsCopied.Should().Be(result.Copy.Kept, "final copied count matches the copy engine"); + finalEc.HeadCommitPosition.Should().BeGreaterThan(0); + finalEc.PercentComplete.Should().BeGreaterThanOrEqualTo(99.9d, "the copy reached the $all head"); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/MigrationRulesTests.cs b/src/MicroPlumberd.Migration.Tests/MigrationRulesTests.cs new file mode 100644 index 0000000..27c6c9b --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/MigrationRulesTests.cs @@ -0,0 +1,167 @@ +using System.Text.Json.Nodes; +using FluentAssertions; +using MicroPlumberd.Migration; +using Xunit; + +namespace MicroPlumberd.Migration.Tests; + +/// Unit tests for the raw rule engine — no EventStore required. +public class MigrationRulesTests +{ + private static EventContext Ctx(string stream, string type, string dataJson = "{}", string? metaJson = null) => + new() + { + SourceStream = stream, + EventNumber = 0, + TargetStream = stream, + Type = type, + Data = JsonNode.Parse(dataJson), + Meta = metaJson is null ? new JsonObject() : JsonNode.Parse(metaJson) + }; + + private static MigrationPlan Plan(params Migration[] migrations) => + new(migrations.Select(CompiledMigration.Compile).ToList()); + + private static EventContext Fold(EventContext ctx, params Migration[] migrations) + { + Plan(migrations).Apply(ctx, null); + return ctx; + } + + [Fact] + public void DropStream_drops_matching_stream_only() + { + Fold(Ctx("Foo-1", "X"), new TestMigrations.DropOneStream()).Dropped.Should().BeTrue(); + Fold(Ctx("Foo-2", "X"), new TestMigrations.DropOneStream()).Dropped.Should().BeFalse(); + } + + [Fact] + public void DropStreams_drops_every_listed_stream() + { + Fold(Ctx("Foo-1", "X"), new TestMigrations.DropManyStreams()).Dropped.Should().BeTrue(); + Fold(Ctx("Foo-2", "X"), new TestMigrations.DropManyStreams()).Dropped.Should().BeTrue(); + Fold(Ctx("Foo-3", "X"), new TestMigrations.DropManyStreams()).Dropped.Should().BeFalse(); + } + + [Fact] + public void DropStream_predicate_drops_by_name_shape() + { + Fold(Ctx("Temp-abc", "X"), new TestMigrations.DropByStreamPredicate()).Dropped.Should().BeTrue(); + Fold(Ctx("Keep-abc", "X"), new TestMigrations.DropByStreamPredicate()).Dropped.Should().BeFalse(); + } + + [Fact] + public void DropEvent_predicate_drops_by_event_view() + { + Fold(Ctx("Foo-1", "Noise"), new TestMigrations.DropByEventPredicate()).Dropped.Should().BeTrue(); + Fold(Ctx("Foo-1", "Signal"), new TestMigrations.DropByEventPredicate()).Dropped.Should().BeFalse(); + } + + [Fact] + public void RenameType_rewrites_the_event_type_string() + { + var ctx = Fold(Ctx("Foo-1", "OldName"), new TestMigrations.RenameTheType()); + ctx.Dropped.Should().BeFalse(); + ctx.Type.Should().Be("NewName"); + } + + [Fact] + public void TransformJson_mutates_payload_and_metadata_in_place() + { + var ctx = Fold(Ctx("Foo-1", "Widget", "{\"a\":1}"), new TestMigrations.TransformData()); + ctx.Data!["a"]!.GetValue().Should().Be(1); + ctx.Data!["added"]!.GetValue().Should().Be("yes"); + ctx.Meta!["touched"]!.GetValue().Should().BeTrue(); + } + + [Fact] + public void TransformJson_does_not_touch_other_types() + { + var ctx = Fold(Ctx("Foo-1", "Gadget", "{\"a\":1}"), new TestMigrations.TransformData()); + (ctx.Data as JsonObject)!.ContainsKey("added").Should().BeFalse(); + } + + [Fact] + public void RenameStream_retargets_events_to_the_new_stream() + { + var ctx = Fold(Ctx("Old-1", "X"), new TestMigrations.RenameTheStream()); + ctx.TargetStream.Should().Be("New-1"); + } + + [Fact] + public void RenameType_then_TransformJson_composes_in_declared_order() + { + var ctx = Fold(Ctx("Foo-1", "V1", "{}"), new TestMigrations.RenameThenTransform()); + ctx.Type.Should().Be("V2"); + ctx.Data!["v2"]!.GetValue().Should().BeTrue(); + } + + [Fact] + public void RenameStream_then_DropStream_composes_in_declared_order() + { + // Renamed to B-1, then the drop that targets B-1 fires. + Fold(Ctx("A-1", "X"), new TestMigrations.RenameStreamThenDrop()).Dropped.Should().BeTrue(); + } + + [Fact] + public void Dropped_event_is_not_transformed_afterwards() + { + // Drop by stream first, then a transform that would otherwise apply — must be skipped. + var drop = new TestMigrations.DropOneStream(); // drops Foo-1 + var transform = new InlineMigration("0099_t", b => + b.TransformJson("Widget", (d, _) => d["should"] = "not-run")); + var ctx = Fold(Ctx("Foo-1", "Widget", "{}"), drop, transform); + ctx.Dropped.Should().BeTrue(); + (ctx.Data as JsonObject)!.ContainsKey("should").Should().BeFalse(); + } + + [Fact] + public void Stats_attribute_effects_to_the_owning_migration() + { + var plan = Plan(new TestMigrations.RenameTheType(), new TestMigrations.DropByEventPredicate()); + + plan.Apply(Ctx("Foo-1", "OldName"), null); // rename fires + plan.Apply(Ctx("Foo-1", "Noise"), null); // drop fires + plan.Apply(Ctx("Foo-1", "OldName"), null); // rename fires again + + plan.Stats["0005_rename_type"].Renamed.Should().Be(2); + plan.Stats["0004_drop_event"].Dropped.Should().Be(1); + } + + [Fact] + public void TransformJson_is_skipped_for_non_json_payload() + { + var ctx = Ctx("Foo-1", "Widget"); + ctx.Data = null; // simulate a non-JSON (binary) payload + Fold(ctx, new TestMigrations.TransformData()); + ctx.Data.Should().BeNull(); // untouched + ctx.Dropped.Should().BeFalse(); // did not crash + } + + [Fact] + public void Plan_requires_payload_parse_only_when_a_rule_needs_it() + { + // DropStreams-only: the payload is never inspected → parse NOTHING (byte-verbatim copy). + var drops = Plan(new TestMigrations.DropManyStreams()); + drops.RequiresPayloadParse("Anything").Should().BeFalse(); + drops.TransformTypes.Should().BeEmpty(); + drops.AnyDropEventRule.Should().BeFalse(); + + // TransformJson is per-TYPE → only that type needs parsing. + var transform = Plan(new TestMigrations.TransformData()); // TransformJson("Widget", …) + transform.TransformTypes.Should().Contain("Widget"); + transform.RequiresPayloadParse("Widget").Should().BeTrue(); + transform.RequiresPayloadParse("Other").Should().BeFalse(); + + // DropEvent hands the predicate PARSED JSON and is opaque → conservatively parse EVERY event. + var dropEvent = Plan(new TestMigrations.DropByEventPredicate()); + dropEvent.AnyDropEventRule.Should().BeTrue(); + dropEvent.RequiresPayloadParse("Anything").Should().BeTrue(); + } + + private sealed class InlineMigration(string id, Action build) : Migration + { + public override string Id => id; + public override void Migrate(IMigrationBuilder b) => build(b); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/ProjectionBehaviorSpikes.cs b/src/MicroPlumberd.Migration.Tests/ProjectionBehaviorSpikes.cs new file mode 100644 index 0000000..8344ef6 --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/ProjectionBehaviorSpikes.cs @@ -0,0 +1,230 @@ +using System.Text; +using System.Text.Json.Nodes; +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Testing; +using Xunit; +using Xunit.Abstractions; + +namespace MicroPlumberd.Migration.Tests; + +/// +/// Exploratory SPIKES (not product code) that settle the [OutputStream] rebuild design against REAL +/// KurrentDB behavior. They answer: (1) does an app-style fromStreams(['$et-A','$et-B']).linkTo(X) +/// catch-up produce ORIGINAL interleaved order or TYPE-CLUSTERED order? (2) can a projection be made to +/// resume at HEAD (ignore history), and what happens if we pre-write the output stream's links? +/// +/// They need Docker (KurrentDB with standard projections enabled — the EventStoreServer fixture enables +/// them). Each spike asserts a HYPOTHESIS so a failure prints the ACTUAL observed order/count, and also +/// writes observations to test output. Run: dotnet test --filter Category=Spike +/// +/// HISTORICAL EVIDENCE (kept after the design was settled): Spike1 demonstrates that an app-style +/// fromStreams(['$et-A','$et-B']).linkTo catch-up TYPE-CLUSTERS (0,2,4,1,3) rather than preserving +/// arrival order — this is exactly why MicroPlumberd's join projection was changed to fromAll() +/// (commit-ordered). Spike2/Spike2c record why the pre-written-link and checkpoint-seed alternatives were +/// rejected. The recovery is proven end-to-end by the SIMPLE + fromAll flow in IncidentReplayIntegrationTests. +/// +[Trait("Category", "Integration")] +[Trait("Category", "Spike")] +public class ProjectionBehaviorSpikes(ITestOutputHelper output) +{ + private sealed class Store : IAsyncDisposable + { + public required EventStoreServer Es { get; init; } + public required KurrentDBClient Client { get; init; } + public required KurrentDBProjectionManagementClient Projections { get; init; } + + public static async Task StartAsync(string tag) + { + var es = EventStoreServer.Create($"mp-spike-{tag}-{Guid.NewGuid():N}"); + await es.StartInDocker(inMemory: true); + var s = es.GetEventStoreSettings(); + return new Store { Es = es, Client = new KurrentDBClient(s), Projections = new KurrentDBProjectionManagementClient(s) }; + } + + public async ValueTask DisposeAsync() => await Es.DisposeAsync(); + } + + private static EventData Typed(string type, int ord) => + new(Uuid.NewUuid(), type, Encoding.UTF8.GetBytes($"{{\"Ord\":{ord}}}")); + + private static EventData Link(ulong eventNumber, string stream) => + new(Uuid.NewUuid(), "$>", Encoding.UTF8.GetBytes($"{eventNumber}@{stream}"), + contentType: "application/octet-stream"); + + // Mirrors MicroPlumberd's projection creation (create -> disable -> update emit:true -> enable). + private async Task CreateEmittingProjectionAsync(Store s, string name, string query) + { + await s.Projections.CreateContinuousAsync(name, query, trackEmittedStreams: true); + await s.Projections.DisableAsync(name); + await s.Projections.UpdateAsync(name, query, emitEnabled: true); + await s.Projections.EnableAsync(name); + } + + private static string JoinQuery(string outputStream, params string[] eventTypes) + { + var streams = string.Join(",", eventTypes.Select(t => $"'$et-{t}'")); + return $"fromStreams([{streams}]).when({{ $any: function(s,e){{ linkTo('{outputStream}', e) }} }});"; + } + + private static async Task> ReadResolvedAsync(KurrentDBClient c, string stream) + { + var res = c.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: true); + var list = new List<(string, int)>(); + if (await res.ReadState == ReadState.StreamNotFound) return list; + await foreach (var re in res) + { + if (re.Event is null) { list.Add(("", -1)); continue; } + var ord = JsonNode.Parse(re.Event.Data.Span)?["Ord"]?.GetValue() ?? -1; + list.Add((re.Event.EventType, ord)); + } + return list; + } + + private static async Task CountAsync(KurrentDBClient c, string stream) + { + var res = c.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: false); + if (await res.ReadState == ReadState.StreamNotFound) return 0; + return await res.LongCountAsync(); + } + + private static async Task WaitUntil(Func> cond, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) { if (await cond()) return; await Task.Delay(500); } + } + + // SPIKE 1 — ordering of a from-beginning fromStreams catch-up. + [Fact] + public async Task Spike1_FromStreams_join_ordering_interleaved_vs_type_clustered() + { + await using var s = await Store.StartAsync("s1"); + + // Interleaved arrival across two event types: A0, B1, A2, B3, A4. + await s.Client.AppendToStreamAsync("Spike-1", StreamState.NoStream, + [Typed("SpikeA", 0), Typed("SpikeB", 1), Typed("SpikeA", 2), Typed("SpikeB", 3), Typed("SpikeA", 4)]); + + // Let $by_event_type populate $et-SpikeA / $et-SpikeB. + await WaitUntil(async () => await CountAsync(s.Client, "$et-SpikeA") >= 3 + && await CountAsync(s.Client, "$et-SpikeB") >= 2, TimeSpan.FromSeconds(30)); + + // App-style join projection (no options()). + await CreateEmittingProjectionAsync(s, "SpikeOut", JoinQuery("SpikeOut", "SpikeA", "SpikeB")); + await WaitUntil(async () => await CountAsync(s.Client, "SpikeOut") >= 5, TimeSpan.FromSeconds(30)); + var appStyle = await ReadResolvedAsync(s.Client, "SpikeOut"); + output.WriteLine("app-style Ord order: " + string.Join(",", appStyle.Select(x => x.Ord))); + + // reorderEvents variant. + var reorderQuery = "options({reorderEvents: true, processingLag: 500});\n" + + JoinQuery("SpikeOutR", "SpikeA", "SpikeB"); + await CreateEmittingProjectionAsync(s, "SpikeOutR", reorderQuery); + await WaitUntil(async () => await CountAsync(s.Client, "SpikeOutR") >= 5, TimeSpan.FromSeconds(30)); + var reordered = await ReadResolvedAsync(s.Client, "SpikeOutR"); + output.WriteLine("reorderEvents Ord order: " + string.Join(",", reordered.Select(x => x.Ord))); + + // SETTLED FINDING (this is WHY MicroPlumberd's join projection was changed to fromAll): a + // from-beginning fromStreams(['$et-A','$et-B']) catch-up over a pre-populated store TYPE-CLUSTERS — it + // drains $et-SpikeA (0,2,4) then $et-SpikeB (1,3), yielding 0,2,4,1,3, NOT the original interleaved + // arrival order 0,1,2,3,4. A fromAll() projection instead processes strictly in $all/commit order and + // reproduces 0,1,2,3,4 (see IncidentReplayIntegrationTests). If a future server build ever made this + // fromStreams catch-up preserve arrival order, it would flip to 0,1,2,3,4 and fail this loudly. + appStyle.Select(x => x.Ord).Should().Equal(new[] { 0, 2, 4, 1, 3 }, + "fromStreams catch-up type-clusters by event type (observed {0}) — this is why the app uses fromAll", + string.Join(",", appStyle.Select(x => x.Ord))); + } + + // SPIKE 2 (now NON-GATING under the live-build approach; kept as evidence) — (baseline) a from-beginning + // projection re-links ALL history; (b) pre-written links duplicate. + [Fact] + public async Task Spike2_from_beginning_relinks_all_history_and_prewritten_links_duplicate() + { + await using var s = await Store.StartAsync("s2"); + + await s.Client.AppendToStreamAsync("Spike2-1", StreamState.NoStream, + [Typed("Spike2A", 0), Typed("Spike2A", 1), Typed("Spike2A", 2)]); + await WaitUntil(async () => await CountAsync(s.Client, "$et-Spike2A") >= 3, TimeSpan.FromSeconds(30)); + + // (baseline) create the emitting projection AFTER history exists → it links ALL 3 (no from-end). + await CreateEmittingProjectionAsync(s, "Spike2Out", JoinQuery("Spike2Out", "Spike2A")); + await WaitUntil(async () => await CountAsync(s.Client, "Spike2Out") >= 3, TimeSpan.FromSeconds(30)); + var baseCount = await CountAsync(s.Client, "Spike2Out"); + output.WriteLine($"from-beginning projection linked {baseCount} history events (expected 3)."); + baseCount.Should().Be(3, "a from-beginning projection re-links ALL history — there is no from-end"); + + // (b) pre-write the output stream's links (as a manual rebuild would), THEN create the emitting + // projection from beginning and see whether EventStore DUPLICATES or reconciles. + await s.Client.AppendToStreamAsync("Spike2Dup", StreamState.NoStream, + [Link(0, "Spike2-1"), Link(1, "Spike2-1"), Link(2, "Spike2-1")]); + await CreateEmittingProjectionAsync(s, "Spike2DupProj", JoinQuery("Spike2Dup", "Spike2A")); + await WaitUntil(async () => await CountAsync(s.Client, "Spike2Dup") >= 4, TimeSpan.FromSeconds(15)); + await Task.Delay(2000); + var dupCount = await CountAsync(s.Client, "Spike2Dup"); + output.WriteLine($"pre-written 3 links + projection from beginning => Spike2Dup now has {dupCount} " + + "(3 = reconciled/skipped; 6 = duplicated)."); + // HYPOTHESIS the rebuild would NEED: reconciliation skips the pre-written links (stays 3). If this + // FAILS with 6, pre-writing + app-registered projection duplicates — the no-double-emit blocker. + dupCount.Should().Be(3, "observed {0} links after regeneration over pre-written links", dupCount); + } + + // SPIKE 2c — EXPLORATORY, best-effort: can seeding the projection checkpoint at head make it ignore + // history? The checkpoint stream format is UNDOCUMENTED, so this is a probe: it reports whether the + // server accepts the write and whether history is skipped. Tolerant of faults by design. + [Fact] + public async Task Spike2c_checkpoint_seed_at_head_exploratory() + { + await using var s = await Store.StartAsync("s2c"); + + await s.Client.AppendToStreamAsync("Spike3-1", StreamState.NoStream, + [Typed("Spike3A", 0), Typed("Spike3A", 1), Typed("Spike3A", 2)]); + await WaitUntil(async () => await CountAsync(s.Client, "$et-Spike3A") >= 3, TimeSpan.FromSeconds(30)); + + const string name = "Spike3Out"; + var query = JoinQuery(name, "Spike3A"); + + // Create disabled, then attempt to seed a checkpoint at head before enabling. + await s.Projections.CreateContinuousAsync(name, query, trackEmittedStreams: true); + await s.Projections.DisableAsync(name); + await s.Projections.UpdateAsync(name, query, emitEnabled: true); + + Position head = Position.Start; + await foreach (var re in s.Client.ReadAllAsync(Direction.Backwards, Position.End, maxCount: 1)) + { head = re.OriginalPosition ?? head; break; } + + var accepted = false; + try + { + // Best-effort guess at a checkpoint payload keyed to the $all position. + var checkpoint = new JsonObject + { + ["$v"] = "1", + ["$checkpointPosition"] = new JsonObject + { + ["commitPosition"] = head.CommitPosition, + ["preparePosition"] = head.PreparePosition + } + }; + await s.Client.AppendToStreamAsync($"$projections-{name}-checkpoint", StreamState.Any, + [new EventData(Uuid.NewUuid(), "$ProjectionCheckpoint", + Encoding.UTF8.GetBytes(checkpoint.ToJsonString()))]); + accepted = true; + } + catch (Exception ex) + { + output.WriteLine($"checkpoint seed write REJECTED: {ex.GetType().Name}: {ex.Message}"); + } + output.WriteLine($"checkpoint seed write accepted by server: {accepted}"); + + await s.Projections.EnableAsync(name); + + // Append a NEW event; if the seed worked, only THIS event links (history ignored). + await s.Client.AppendToStreamAsync("Spike3-1", StreamState.Any, [Typed("Spike3A", 99)]); + await WaitUntil(async () => await CountAsync(s.Client, name) >= 1, TimeSpan.FromSeconds(15)); + await Task.Delay(2000); + + var linked = await ReadResolvedAsync(s.Client, name); + output.WriteLine("linked Ord values: " + string.Join(",", linked.Select(x => x.Ord))); + output.WriteLine($"HISTORY IGNORED (seed worked) == only [99]? {linked.Count == 1 && linked[0].Ord == 99}"); + // No hard assertion: this probe REPORTS. The design decision reads the output above. + linked.Should().NotBeEmpty("the projection should link at least the new event once enabled"); + } +} diff --git a/src/MicroPlumberd.Migration.Tests/TestMigrations.cs b/src/MicroPlumberd.Migration.Tests/TestMigrations.cs new file mode 100644 index 0000000..a7998e6 --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/TestMigrations.cs @@ -0,0 +1,79 @@ +using System.Text.Json.Nodes; +using MicroPlumberd.Migration; + +namespace MicroPlumberd.Migration.Tests; + +/// Inline migrations used by the unit tests. Each declares a small, deterministic rule set. +internal static class TestMigrations +{ + public sealed class DropOneStream : Migration + { + public override string Id => "0001_drop_one"; + public override void Migrate(IMigrationBuilder b) => b.DropStream("Foo-1"); + } + + public sealed class DropManyStreams : Migration + { + public override string Id => "0002_drop_many"; + public override void Migrate(IMigrationBuilder b) => b.DropStreams("Foo-1", "Foo-2"); + } + + public sealed class DropByStreamPredicate : Migration + { + public override string Id => "0003_drop_pred"; + public override void Migrate(IMigrationBuilder b) => + b.DropStream(s => s.StartsWith("Temp-", StringComparison.Ordinal)); + } + + public sealed class DropByEventPredicate : Migration + { + public override string Id => "0004_drop_event"; + public override void Migrate(IMigrationBuilder b) => + b.DropEvent(e => e.Type == "Noise"); + } + + public sealed class RenameTheType : Migration + { + public override string Id => "0005_rename_type"; + public override void Migrate(IMigrationBuilder b) => b.RenameType("OldName", "NewName"); + } + + public sealed class TransformData : Migration + { + public override string Id => "0006_transform"; + public override void Migrate(IMigrationBuilder b) => + b.TransformJson("Widget", (data, meta) => + { + data["added"] = "yes"; + meta["touched"] = true; + }); + } + + public sealed class RenameTheStream : Migration + { + public override string Id => "0007_rename_stream"; + public override void Migrate(IMigrationBuilder b) => b.RenameStream("Old-1", "New-1"); + } + + /// Rename type then transform the NEW type — composition across ops in declared order. + public sealed class RenameThenTransform : Migration + { + public override string Id => "0008_rename_then_transform"; + public override void Migrate(IMigrationBuilder b) + { + b.RenameType("V1", "V2"); + b.TransformJson("V2", (data, _) => data["v2"] = true); + } + } + + /// Rename a stream then drop the NEW stream name — composition of stream ops. + public sealed class RenameStreamThenDrop : Migration + { + public override string Id => "0009_rename_stream_then_drop"; + public override void Migrate(IMigrationBuilder b) + { + b.RenameStream("A-1", "B-1"); + b.DropStream("B-1"); + } + } +} diff --git a/src/MicroPlumberd.Migration.Tests/UserDefinedIndexIntegrationTests.cs b/src/MicroPlumberd.Migration.Tests/UserDefinedIndexIntegrationTests.cs new file mode 100644 index 0000000..04eb7d4 --- /dev/null +++ b/src/MicroPlumberd.Migration.Tests/UserDefinedIndexIntegrationTests.cs @@ -0,0 +1,315 @@ +using System.Text; +using System.Text.Json.Nodes; +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Testing; +using MicroPlumberd.Tests.App.Domain; +using Xunit; +using Xunit.Abstractions; + +namespace MicroPlumberd.Migration.Tests; + +/// +/// Integration proof of the CLEAN merge read-path against a REAL KurrentDB 26.1 (user-defined indexes), the +/// alternative to the paced . Uses the same Docker +/// fixture as the other migration tests (NEVER Testcontainers). +/// +/// Covers: (a) create + ready-poll (the STARTED-immediately / backfill-async gotcha is handled); +/// (b) an interleaved multi-type merge reads in COMMIT order (0,1,2,3,4), NOT the type-clustered 0,2,4,1,3 that +/// a fromStreams join produces (proven in ); (c) idempotent +/// re-create; (d) a full migration with a copies the aggregate +/// streams faithfully AND its dest index reproduces the SAME merged order the +/// path builds into the physical merge stream. +/// +[Trait("Category", "Integration")] +public class UserDefinedIndexIntegrationTests(ITestOutputHelper output) +{ + private sealed class Store : IAsyncDisposable + { + private readonly List _servers = new(); + + public async Task<(KurrentDBClient Client, KurrentDBProjectionManagementClient Projections, string Conn, IPlumber Plumber)> + NewAsync(string tag) + { + var es = EventStoreServer.Create($"mp-udix-{tag}-{Guid.NewGuid():N}"); + _servers.Add(es); + await es.StartInDocker(inMemory: true); + var s = es.GetEventStoreSettings(); + return (new KurrentDBClient(s), new KurrentDBProjectionManagementClient(s), es.HttpUrl.ToString(), + global::MicroPlumberd.Plumber.Create(s)); + } + + public async ValueTask DisposeAsync() + { + foreach (var s in _servers) await s.DisposeAsync(); + } + } + + private static EventData Typed(string type, int ord) => + new(Uuid.NewUuid(), type, Encoding.UTF8.GetBytes($"{{\"Ord\":{ord}}}")); + + private static async Task> ReadOrdsAsync(UserDefinedIndexSource src, string name, CancellationToken ct = default) + { + var ords = new List(); + await foreach (var e in src.ReadAsync(name, ct)) + ords.Add(e.Data?["Ord"]?.GetValue() ?? -1); + return ords; + } + + // (a) + (b): interleaved arrival across two event types must read back in COMMIT order via the index — the + // core property. Also exercises create + WaitUntilReady (handles STARTED-immediately + async backfill). + [Fact] + public async Task Index_reads_interleaved_multitype_merge_in_commit_order_not_type_clustered() + { + await using var stores = new Store(); + var (client, _, conn, _) = await stores.NewAsync("order"); + + // A0, B1, A2, B3, A4 — a fromStreams join would type-cluster this to 0,2,4,1,3 (ProjectionBehaviorSpikes). + await client.AppendToStreamAsync("Spike-1", StreamState.NoStream, + [Typed("UdiA", 0), Typed("UdiB", 1), Typed("UdiA", 2), Typed("UdiB", 3), Typed("UdiA", 4)]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec + { + Name = "merge-ab", + EventTypes = new HashSet { "UdiA", "UdiB" } + }; + + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 5, TimeSpan.FromSeconds(30)); + var ords = await ReadOrdsAsync(src, spec.Name); + + output.WriteLine("index Ord order: " + string.Join(",", ords)); + ords.Should().Equal(new[] { 0, 1, 2, 3, 4 }, + "the user-defined index is read via filtered $all → commit order, NOT the type-clustered 0,2,4,1,3"); + ords.Should().NotEqual(new[] { 0, 2, 4, 1, 3 }, "commit order must not degrade to type-clustered"); + } + + // (a) explicit: ready-polling completes and the read is COMPLETE (all backfilled) even when called + // immediately after creation (the index reports STARTED before it has finished backfilling history). + [Fact] + public async Task Create_then_immediately_wait_ready_yields_complete_backfilled_read() + { + await using var stores = new Store(); + var (client, _, conn, _) = await stores.NewAsync("ready"); + + const int n = 50; + var batch = Enumerable.Range(0, n).Select(i => Typed(i % 2 == 0 ? "UdiA" : "UdiB", i)).ToArray(); + await client.AppendToStreamAsync("Big-1", StreamState.NoStream, batch); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec + { + Name = "readytest", + EventTypes = new HashSet { "UdiA", "UdiB" } + }; + + // Ensure + wait-ready back-to-back: WaitUntilReadyAsync must not return before the backfill is complete. + // The expected count is computed AUTHORITATIVELY from the store's own $all (not hard-coded), then required. + await src.EnsureAsync(spec); + var expected = await src.CountMatchingAsync(spec.EventTypes); + expected.Should().Be(n, "CountMatchingAsync reads the exact number of matching events from $all"); + await src.WaitUntilReadyAsync(spec.Name, expected, TimeSpan.FromSeconds(30)); + + var ords = await ReadOrdsAsync(src, spec.Name); + ords.Should().HaveCount(n, "ready-poll must wait for the full backfill, not return at STARTED"); + ords.Should().Equal(Enumerable.Range(0, n), "and the backfilled read is in commit order"); + } + + // (c): re-creating the same index (same name + filter) is a no-op and does not disturb the read. + [Fact] + public async Task Ensure_is_idempotent_on_recreate() + { + await using var stores = new Store(); + var (client, _, conn, _) = await stores.NewAsync("idem"); + await client.AppendToStreamAsync("S-1", StreamState.NoStream, [Typed("UdiA", 0), Typed("UdiB", 1)]); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec + { + Name = "idempotent", + EventTypes = new HashSet { "UdiA", "UdiB" } + }; + + await src.EnsureAsync(spec); + await src.WaitUntilReadyAsync(spec.Name, expectedCount: 2, TimeSpan.FromSeconds(30)); + + // Second Ensure with the identical spec must NOT throw (409 already-exists is treated as success). + var act = async () => await src.EnsureAsync(spec); + await act.Should().NotThrowAsync(); + + var ords = await ReadOrdsAsync(src, spec.Name); + ords.Should().Equal(new[] { 0, 1 }, "the index is unchanged after an idempotent re-create"); + } + + // PROVING the readiness is AUTHORITATIVE, not a stability heuristic: it blocks until the indexed count + // equals the KNOWN total and FAILS LOUD (TimeoutException) if that total is never reached — a heuristic + // would instead have "settled" on whatever count two consecutive polls happened to agree on. + [Fact] + public async Task WaitUntilReady_is_authoritative_waits_for_exact_count_and_times_out_if_short() + { + await using var stores = new Store(); + var (client, _, conn, _) = await stores.NewAsync("auth"); + + const int n = 40; + var batch = Enumerable.Range(0, n).Select(i => Typed(i % 2 == 0 ? "UdiA" : "UdiB", i)).ToArray(); + await client.AppendToStreamAsync("Auth-1", StreamState.NoStream, batch); + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec + { + Name = "authoritative", + EventTypes = new HashSet { "UdiA", "UdiB" } + }; + await src.EnsureAsync(spec); + + // The authoritative expected count comes from the store's own $all — exactly n. + (await src.CountMatchingAsync(spec.EventTypes)).Should().Be(n); + + // Waiting for the REAL total succeeds and yields exactly n, in commit order. + await src.WaitUntilReadyAsync(spec.Name, expectedCount: n, TimeSpan.FromSeconds(30)); + (await ReadOrdsAsync(src, spec.Name)).Should().Equal(Enumerable.Range(0, n)); + + // Waiting for MORE than exist can NEVER be satisfied → it times out rather than falsely reporting ready. + // (A stability heuristic would have returned at n; the authoritative wait refuses to.) + var act = async () => await src.WaitUntilReadyAsync(spec.Name, expectedCount: n + 1, TimeSpan.FromSeconds(4)); + (await act.Should().ThrowAsync()) + .Which.Message.Should().Contain($"{n + 1}", "the timeout must report the unreachable target count"); + } + + // PROVING no truncation on a LARGE merge that spans many backfill/poll intervals: a heuristic gate ("two + // equal counts 100ms apart") could false-ready on a mid-backfill lull and yield a truncated tail; the + // authoritative expected-count gate must yield ALL n, in commit order, no matter how the backfill paces. + // n is large enough that the backfill is NOT instantaneous (thousands of links across many $et catch-ups). + [Fact] + public async Task Large_merge_reads_back_full_count_in_commit_order_no_truncation() + { + await using var stores = new Store(); + var (client, _, conn, _) = await stores.NewAsync("large"); + + const int n = 20_000; + // Interleaved across two event types so a fromStreams join would type-cluster; appended in chunks under + // KurrentDB's per-append cap, preserving global Ord order 0..n-1 in $all commit order. + var expected = StreamState.NoStream; + for (var offset = 0; offset < n; offset += 1000) + { + var chunk = Enumerable.Range(offset, Math.Min(1000, n - offset)) + .Select(i => Typed(i % 2 == 0 ? "UdiA" : "UdiB", i)).ToArray(); + await client.AppendToStreamAsync("Large-1", expected, chunk); + expected = (ulong)(offset + chunk.Length - 1); + } + + var src = new UserDefinedIndexSource(client, conn); + var spec = new UserDefinedIndexSpec { Name = "large-merge", EventTypes = new HashSet { "UdiA", "UdiB" } }; + await src.EnsureAsync(spec); + + var total = await src.CountMatchingAsync(spec.EventTypes); + total.Should().Be(n, "the authoritative expected count is the exact number of matching events in $all"); + + await src.WaitUntilReadyAsync(spec.Name, total, TimeSpan.FromSeconds(120)); + + // The whole merge reads back — full count AND commit order 0..n-1, i.e. NOT a truncated tail and NOT + // type-clustered. This is the property the expected-count gate guarantees regardless of backfill pacing. + var ords = await ReadOrdsAsync(src, spec.Name); + ords.Should().HaveCount(n, "the reader must yield every indexed event, never a truncated tail"); + ords.Should().Equal(Enumerable.Range(0, n), "in true commit order, not type-clustered"); + } + + // (d): a full migration using the index copy context — aggregate streams copied faithfully (verification OK), + // and the DEST index reproduces the SAME merged commit order the projection-copy path builds into + // FooModel_v1. Cross-checks the two merge-recovery strategies against ONE source. + [Fact] + public async Task Index_copy_matches_projection_copy_merged_order_and_faithful_aggregate_copy() + { + await using var stores = new Store(); + var (srcClient, srcProj, srcConn, srcPlumber) = await stores.NewAsync("dsrc"); + + // Interleaved arrival across two aggregates: FooCreated(a), FooRefined(a), FooCreated(b), FooRefined(b). + var a = Guid.NewGuid(); + var b = Guid.NewGuid(); + await srcPlumber.SaveNew(FooAggregate.Open("a-created", a)); + var aggA = await srcPlumber.Get(a); + aggA.Refine("a-refined"); await srcPlumber.SaveChanges(aggA); + await srcPlumber.SaveNew(FooAggregate.Open("b-created", b)); + var aggB = await srcPlumber.Get(b); + aggB.Refine("b-refined"); await srcPlumber.SaveChanges(aggB); + + // Build the app's [OutputStream] join projection on source so the index copy can DISCOVER it. + await srcPlumber.TryCreateJoinProjection(); + await WaitUntil(async () => await CountStreamAsync(srcClient, "FooModel_v1") >= 4, TimeSpan.FromSeconds(30)); + + var expected = new[] { "a-created", "a-refined", "b-created", "b-refined" }; + + // --- INDEX COPY path --- + var (idxClient, idxProj, idxConn, _) = await stores.NewAsync("didx"); + var indexResult = await new MigrationRunner().RunAsync(srcClient, idxClient, Array.Empty(), + dryRun: false, projectionCopy: null, + indexCopy: new UserDefinedIndexCopyContext + { + SourceProjections = srcProj, + SourceConnectionString = srcConn, + DestConnectionString = idxConn, + ReadyTimeout = TimeSpan.FromSeconds(30) + }); + + indexResult.Verification!.AllOk.Should().BeTrue(indexResult.Verification.Format()); + var created = indexResult.CreatedIndexes.Should().ContainSingle(i => i.OutputStream == "FooModel_v1").Subject; + created.EventTypes.Should().BeEquivalentTo(new[] { "FooCreated", "FooRefined" }); + + var idxSource = new UserDefinedIndexSource(idxClient, idxConn); + var names = new List(); + await foreach (var e in idxSource.ReadAsync(created.IndexName)) + names.Add(e.Data?["Name"]?.GetValue()); + output.WriteLine("index merged order: " + string.Join(",", names)); + names.Should().Equal(expected, "the index reproduces the merge in commit order, not type-clustered"); + + // --- PROJECTION COPY path (cross-check) — same source → the physical FooModel_v1 must match --- + var (projClient, projProj, _, _) = await stores.NewAsync("dproj"); + await new MigrationRunner().RunAsync(srcClient, projClient, Array.Empty(), dryRun: false, + projectionCopy: new ProjectionCopyContext + { + SourceProjections = srcProj, + DestProjections = projProj, + SourceConnectionString = srcConn, + PerEventPaceTimeout = TimeSpan.FromSeconds(30), + DrainTimeout = TimeSpan.FromSeconds(60) + }); + await WaitUntil(async () => await CountStreamAsync(projClient, "FooModel_v1") >= 4, TimeSpan.FromSeconds(30)); + var projNames = await ReadJoinNamesAsync(projClient, "FooModel_v1"); + projNames.Should().Equal(expected); + + // The two strategies agree on the merged order — the index is a faithful alternative to the paced copy. + names.Should().Equal(projNames, "index-copy and projection-copy must produce the same merged order"); + } + + private static async Task CountStreamAsync(KurrentDBClient c, string stream) + { + var res = c.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: false); + if (await res.ReadState == ReadState.StreamNotFound) return 0; + return await res.LongCountAsync(); + } + + private static async Task> ReadJoinNamesAsync(KurrentDBClient c, string stream) + { + var res = c.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: true); + var names = new List(); + if (await res.ReadState == ReadState.StreamNotFound) return names; + await foreach (var re in res) + { + if (re.Event is null) { names.Add(null); continue; } + names.Add(JsonNode.Parse(re.Event.Data.Span)?["Name"]?.GetValue()); + } + return names; + } + + private static async Task WaitUntil(Func> condition, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + if (await condition()) return; + await Task.Delay(500); + } + } +} diff --git a/src/MicroPlumberd.Migration/CopyEngine.cs b/src/MicroPlumberd.Migration/CopyEngine.cs new file mode 100644 index 0000000..d21b896 --- /dev/null +++ b/src/MicroPlumberd.Migration/CopyEngine.cs @@ -0,0 +1,500 @@ +using System.Buffers; +using System.Security.Cryptography; +using System.Text; +using System.Text.Json; +using System.Text.Json.Nodes; +using KurrentDB.Client; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MicroPlumberd.Migration; + +/// Per source-stream copy accounting. +public sealed class StreamCopyInfo +{ + /// The destination stream these events were written to (after any rename). + public required string TargetStream { get; set; } + + /// Events read from this source stream (excludes system/unparseable events). + public long SourceCount { get; set; } + + /// Events from this source stream written to the destination. + public long Kept { get; set; } + + /// Events from this source stream dropped by migration rules. + public long Dropped { get; set; } +} + +/// The outcome of a copy (or dry) run. +public sealed class CopyResult +{ + /// True when the run wrote nothing to the destination. + public required bool DryRun { get; init; } + + /// The UTC instant stamped as MigratedAt on every copied event in this run. + public required DateTime RunTimeUtc { get; init; } + + /// Total user events scanned from the source (system/`$` streams excluded). + public long SourceEvents { get; internal set; } + + /// Total events written to the destination (0 on a dry run). + public long Kept { get; internal set; } + + /// Total events dropped by migration rules. + public long Dropped { get; internal set; } + + /// + /// Events whose payload was declared JSON but could not be parsed and were therefore COPIED VERBATIM + /// (byte-for-byte) rather than dropped — a warning count, NOT data loss. Their bytes are preserved exactly. + /// + public long UnparseableVerbatim { get; internal set; } + + /// + /// Link events ($>) skipped in non-system streams — i.e. [OutputStream] merge/link + /// streams. These are projection output (built by continuous linkTo projections the app + /// re-registers on the destination), so the destination regenerates them from the migrated aggregate + /// streams; copying them would duplicate links once the projection re-runs. + /// + public long LinkEventsSkipped { get; internal set; } + + /// Per-migration effect counters. + public required IReadOnlyDictionary MigrationStats { get; init; } + + /// Per source-stream accounting, keyed by source stream id. + public Dictionary SourceStreams { get; } = new(StringComparer.Ordinal); + + /// + /// Final revision (last event number) per destination stream. Computed on a dry run too (the append is + /// skipped, but the revision bookkeeping still runs so counts/verification can be reported). + /// + public Dictionary DestFinalRevision { get; } = new(StringComparer.Ordinal); + + /// + /// Number of batched append operations performed against the destination (each a single + /// AppendToStreamAsync of a non-empty batch). More than one per stream means the batch caps forced a + /// SPLIT — a large/bulk stream was written across several appends. Counted on a dry run too (it reflects the + /// appends that WOULD be issued). + /// + public long DestAppendCount { get; internal set; } + + /// + /// Write-fidelity checksum per destination stream: the hex SHA-256 of every written event's + /// (EventType UTF-8 || 0x00 || Data bytes), folded in write order. The verifier re-reads each dest + /// stream and recomputes the same hash to catch dest-side REORDER / TRUNCATION / CORRUPTION that a count + /// check cannot. Empty on a dry run (nothing was written). This checks the copy INTENT reached the dest + /// faithfully — NOT source-re-derivation or transform semantics (those are covered by tests). + /// + public Dictionary DestStreamHashes { get; } = new(StringComparer.Ordinal); +} + +/// +/// The offline copy engine: reads a SOURCE store in $all commit order, applies the pending +/// migrations' raw rules to each event, and writes the surviving events to a fresh DESTINATION store, +/// per stream, preserving per-stream order and (gapless) version. +/// +/// +/// Skipping. Every $-prefixed stream is skipped ($ce-*, $et-*, $streams, $scavenges, +/// projection/stats streams, stream-metadata `$$…`) — the destination regenerates its own system +/// projections — as is any explicitly reserved stream (the migration-history stream). Events whose +/// type starts with $ (e.g. +/// $streamDeleted, $metadata, link events) are skipped too, which is how tombstone / +/// StreamDeleted markers are handled without crashing. Events whose payload is declared JSON but cannot be +/// parsed are NEVER dropped — they are copied VERBATIM (byte-for-byte) and counted as +/// . +/// Tombstoned source streams. Reading $all with resolveLinkTos:false never +/// dereferences the null link-targets a tombstoned stream leaves in its $ce category stream, so a +/// source that already contains hard-tombstoned streams is copied without error. +/// Streaming. The source log is streamed, never materialised. Writes are batched only across +/// events that are adjacent in $all and target the same destination stream (i.e. same-commit writes), +/// which keeps order and version exact. +/// +internal sealed class CopyEngine(ILogger? logger = null) : IMigrationOpLog +{ + private readonly ILogger _logger = logger ?? NullLogger.Instance; + + /// KurrentDB's link-event type, emitted by linkTo projections into merge streams. + private const string LinkEventType = "$>"; + + /// + /// Max events per single AppendToStreamAsync. Batches are capped so one bulk/dominant stream (or many + /// streams renamed into one) never builds an oversized append that exceeds KurrentDB's per-append limits. + /// + private const int MaxBatchCount = 1000; + + /// + /// Max cumulative payload bytes per append — a conservative fraction of KurrentDB's MaxAppendSize (~16 MiB). + /// The batch flushes BEFORE adding an event that would exceed this (or ). A single + /// event larger than this is appended alone. + /// + private const int MaxBatchBytes = 1 * 1024 * 1024; + + /// Rough per-event framing overhead (id/type/headers) added to payload size for the byte cap. + private const int PerEventFramingOverheadBytes = 128; + + private static readonly byte[] HashSeparator = [0x00]; + + public async Task RunAsync( + KurrentDBClient source, + KurrentDBClient dest, + MigrationPlan plan, + DateTime runTimeUtc, + bool dryRun, + IReadOnlySet reservedStreams, + MigrationProgressTracker? progress = null, + IEventPacer? pacer = null, + CancellationToken ct = default) + { + var result = new CopyResult + { + DryRun = dryRun, + RunTimeUtc = runTimeUtc, + MigrationStats = plan.Stats + }; + + // Pre-flight: a real run REQUIRES a fresh, empty destination (every first append expects NoStream). + // Fail up front with a clear message instead of a raw WrongExpectedVersion mid-copy. + if (!dryRun) + await EnsureDestinationEmptyAsync(dest, ct).ConfigureAwait(false); + + // Capture the LAST COPYABLE event's $all commit position up front so event-copy progress is a real + // percentage that reaches 100% (position-based; counting every event would need a second full $all + // scan). It must exclude the trailing $-system / $> merge-link events the copy skips — otherwise the + // copy's last processed position sits before the raw $all head and the percent never reaches 100. + if (progress is not null) + progress.EventCopyPlan(await ReadLastCopyablePositionAsync(source, reservedStreams, ct) + .ConfigureAwait(false)); + + // Reusable buffer + JSON writer. The metadata MigratedAt re-stamp runs for EVERY copied event (it is + // inherent to that feature — one JsonDocument parse + one output byte[] per event); the event payload is + // additionally re-serialised only for the transformed minority. Reusing one writer (Reset per use) + // avoids a per-event intermediate string (ToJsonString) and buffer re-allocation. The copy loop is + // single-threaded, so a single writer + a single reused EventContext are safe. + var scratch = new ArrayBufferWriter(1024); + using var writer = new Utf8JsonWriter(scratch); + var ctx = new EventContext + { + SourceStream = "", EventNumber = 0, TargetStream = "", Type = "", + EventId = Guid.Empty, Created = default + }; + + // Per-dest-stream rolling SHA-256 over each written event (write-fidelity checksum for the verifier). + var hashers = new Dictionary(StringComparer.Ordinal); + + // Pending batch of adjacent same-target-stream events, capped by count AND cumulative bytes (M1). + string? batchStream = null; + var batch = new List(); + var batchBytes = 0L; + + async Task FlushAsync() + { + if (batchStream is null || batch.Count == 0) return; + + StreamState expected; + ulong newLast; + if (result.DestFinalRevision.TryGetValue(batchStream, out var last)) + { + expected = last; // implicit ulong -> StreamRevision + newLast = last + (ulong)batch.Count; + } + else + { + expected = StreamState.NoStream; + newLast = (ulong)(batch.Count - 1); + } + + if (!dryRun) + await dest.AppendToStreamAsync(batchStream, expected, batch, cancellationToken: ct) + .ConfigureAwait(false); + + result.DestAppendCount++; // one append op (a split shows up as >1 per stream) + result.DestFinalRevision[batchStream] = newLast; // supports multiple flushes to the SAME stream (gapless) + batch.Clear(); + batchBytes = 0; + batchStream = null; + } + + // Folds a written event into its dest-stream rolling hash: SHA256(EventType || 0x00 || Data), in order. + void FeedHash(string destStream, EventData ed) + { + if (!hashers.TryGetValue(destStream, out var h)) + hashers[destStream] = h = IncrementalHash.CreateHash(HashAlgorithmName.SHA256); + var type = ed.Type; + var max = Encoding.UTF8.GetMaxByteCount(type.Length); + byte[]? rented = max > 512 ? ArrayPool.Shared.Rent(max) : null; + Span buf = rented ?? stackalloc byte[512]; + var n = Encoding.UTF8.GetBytes(type, buf); + h.AppendData(buf[..n]); + if (rented is not null) ArrayPool.Shared.Return(rented); + h.AppendData(HashSeparator); + h.AppendData(ed.Data.Span); + } + + var read = source.ReadAllAsync(Direction.Forwards, Position.Start, resolveLinkTos: false, + cancellationToken: ct); + + await foreach (var re in read.ConfigureAwait(false)) + { + // Distinguish the three stream classes: + // 1. system streams ($ce-*, $et-*, $streams, $$meta, projections, stats, $scavenges) — SKIP; + // KurrentDB regenerates them on the destination. + // 2. [OutputStream] merge/link streams — their events are $> links. SKIP + count; they are + // projection output and the destination's re-registered linkTo projections rebuild them + // from the migrated aggregate streams (copying would duplicate links). + // 3. regular aggregate/domain streams — copied raw through the rule engine (below). + var er = re.Event; + if (er is null) continue; // unresolved artifact + if (er.EventStreamId.Length > 0 && er.EventStreamId[0] == '$') continue; // (1) system stream + if (reservedStreams.Contains(er.EventStreamId)) continue; // reserved (history) + if (er.EventType == LinkEventType) // (2) merge/link-stream link + { + result.LinkEventsSkipped++; + continue; + } + if (er.EventType.Length > 0 && er.EventType[0] == '$') continue; // other system/tombstone event + + var sourceStream = er.EventStreamId; + + // PLAN-DRIVEN PARSE: parse the payload ONLY when a rule needs it — a TransformJson targets this + // type, OR any DropEvent rule exists (its predicate is handed PARSED JSON). The common case (e.g. + // _0001 is DropStreams-only) parses NOTHING here and copies the event byte-verbatim below. + var parse = plan.RequiresPayloadParse(er.EventType); + JsonNode? data = null; + JsonNode? meta = null; + if (parse) + { + data = TryParseJson(er.Data); + if (data is null && !er.Data.IsEmpty && IsJsonContent(er.ContentType)) + { + // Declared JSON but unparseable. Do NOT drop it (M2) — copy it VERBATIM. Data stays null so + // BuildEventData writes the ORIGINAL bytes unchanged; a DropEvent predicate still sees null + // and may choose to drop; TransformJson skips a null payload. Never lose an event to a parse + // failure the migration didn't ask to act on. + _logger.LogWarning("Unparseable JSON payload for {Type} in {Stream}#{Num} — copied VERBATIM " + + "(not dropped).", er.EventType, sourceStream, er.EventNumber.ToUInt64()); + result.UnparseableVerbatim++; + } + meta = TryParseJson(er.Metadata); // metadata may be valid even when the payload isn't + } + + var info = GetOrAdd(result, sourceStream); + info.SourceCount++; + result.SourceEvents++; + + // Reuse ONE EventContext across the loop (m5) — it never escapes except into the synchronous + // DropEvent predicate, which takes an immutable AsRawEvent() snapshot. + ctx.SourceStream = sourceStream; + ctx.EventNumber = er.EventNumber.ToUInt64(); + ctx.TargetStream = sourceStream; + ctx.Type = er.EventType; + ctx.Data = data; + ctx.Meta = meta; + ctx.Dropped = false; + ctx.Transformed = false; + // Identity fields rules may MATCH on but never rewrite: the copy preserves the source event id + // (BuildEventData below) and the destination stamps its own write timestamp. + ctx.EventId = er.EventId.ToGuid(); + ctx.Created = er.Created; + + plan.Apply(ctx, this); + + if (ctx.Dropped) + { + info.Dropped++; + result.Dropped++; + // Advance progress by position even for dropped events, so the percentage still reaches 100 + // if the last copyable event happens to be dropped by a rule. + progress?.EventCopied(result.Kept, sourceStream, er.Position.CommitPosition); + continue; + } + + // Retarget accounting to the (possibly renamed) destination stream. + info.TargetStream = ctx.TargetStream; + info.Kept++; + result.Kept++; + + var ed = BuildEventData(er, ctx, runTimeUtc, scratch, writer); + var edSize = ed.Data.Length + ed.Metadata.Length + PerEventFramingOverheadBytes; + + // Flush BEFORE adding when the target stream changes OR the batch would exceed either cap (M1). A + // same-stream cap-flush is safe: DestFinalRevision continues the expected version → gapless. + if (batchStream is not null && + (batchStream != ctx.TargetStream || batch.Count >= MaxBatchCount || batchBytes + edSize > MaxBatchBytes)) + await FlushAsync().ConfigureAwait(false); + batchStream ??= ctx.TargetStream; + batch.Add(ed); + batchBytes += edSize; + if (!dryRun) FeedHash(ctx.TargetStream, ed); // write-fidelity checksum (real run only) + + // PACED mode (projection copy): flush THIS event immediately and wait for the join projection(s) to + // emit its link before the next write — so the projection is kept caught up event-by-event and can + // never accumulate a backlog (the only thing a fromStreams catch-up type-clusters). No-op on dry run. + if (pacer is not null && !dryRun) + { + await FlushAsync().ConfigureAwait(false); + await pacer.AfterAppendedAsync(ctx.Type, ct).ConfigureAwait(false); + } + + // Report progress (throttled inside the tracker): count written + current stream + $all position. + progress?.EventCopied(result.Kept, sourceStream, er.Position.CommitPosition); + } + + await FlushAsync().ConfigureAwait(false); + + // Finalise the per-dest-stream write-fidelity hashes for the verifier. + foreach (var (stream, h) in hashers) + { + result.DestStreamHashes[stream] = Convert.ToHexString(h.GetHashAndReset()); + h.Dispose(); + } + + _logger.LogInformation( + "{Mode}: scanned {Source} user event(s); kept {Kept}, dropped {Dropped}, unparseable-verbatim {Verbatim}, " + + "merge-link events skipped {Links} (regenerated by dest projections).", + dryRun ? "Dry run" : "Copy", result.SourceEvents, result.Kept, result.Dropped, result.UnparseableVerbatim, + result.LinkEventsSkipped); + return result; + } + + void IMigrationOpLog.TransformSkippedNonJson(string type, string stream, ulong eventNumber) => + _logger.LogWarning("TransformJson('{Type}') skipped for non-JSON payload in {Stream}#{Num}.", + type, stream, eventNumber); + + // The $all commit position of the last COPYABLE event — the last event the copy will actually append, so + // the copy reaches 100% when it processes it. Scans $all backwards, skipping the same classes the copy + // skips ($-streams, reserved streams, $>/$-typed events), and returns the first match (0 if none). + private static async Task ReadLastCopyablePositionAsync(KurrentDBClient source, + IReadOnlySet reservedStreams, CancellationToken ct) + { + await foreach (var re in source + .ReadAllAsync(Direction.Backwards, Position.End, resolveLinkTos: false, + cancellationToken: ct).ConfigureAwait(false)) + { + var er = re.Event; + if (er is null) continue; + if (er.EventStreamId.Length > 0 && er.EventStreamId[0] == '$') continue; + if (reservedStreams.Contains(er.EventStreamId)) continue; + if (er.EventType == LinkEventType) continue; + if (er.EventType.Length > 0 && er.EventType[0] == '$') continue; + return er.Position.CommitPosition; + } + return 0; + } + + private static StreamCopyInfo GetOrAdd(CopyResult result, string sourceStream) + { + if (!result.SourceStreams.TryGetValue(sourceStream, out var info)) + result.SourceStreams[sourceStream] = info = new StreamCopyInfo { TargetStream = sourceStream }; + return info; + } + + // Fails fast unless the destination has no user (non-$) events. A truly fresh node still has system + // streams ($stats/$scavenges/standard-projection state), so only non-$ streams signal "not empty". + private static async Task EnsureDestinationEmptyAsync(KurrentDBClient dest, CancellationToken ct) + { + await foreach (var re in dest.ReadAllAsync(Direction.Forwards, Position.Start, cancellationToken: ct) + .ConfigureAwait(false)) + { + var er = re.Event; + if (er is null) continue; + if (er.EventStreamId.Length > 0 && er.EventStreamId[0] == '$') continue; + throw new InvalidOperationException( + $"Destination store is not empty: found stream '{er.EventStreamId}'. The migration requires " + + "a FRESH, empty destination (start a new EventStore on a clean volume)."); + } + } + + private static EventData BuildEventData(EventRecord er, EventContext ctx, DateTime runTimeUtc, + ArrayBufferWriter scratch, Utf8JsonWriter writer) + { + // DATA: re-serialise ONLY when a TransformJson actually mutated the payload; otherwise copy the + // original bytes VERBATIM (no parse, no string, no copy). A stream/type rename never touches the + // payload, so those events still copy verbatim. + ReadOnlyMemory dataBytes = ctx is { Transformed: true, Data: not null } + ? SerializeNode(ctx.Data, scratch, writer) + : er.Data; + + // METADATA: stamp MigratedAt. When a transform mutated the metadata, re-serialise the mutated node — + // stamping MigratedAt when it is an object, or writing the mutated non-object node as-is (m6: never + // silently discard a transform's metadata mutation by falling back to the original bytes). Otherwise + // stamp the ORIGINAL metadata bytes — object → copy props + append MigratedAt via the pooled writer (no + // intermediate string); non-object/non-JSON → verbatim; empty → a fresh object. + ReadOnlyMemory metaBytes = ctx switch + { + { Transformed: true, Meta: JsonObject metaObj } => SerializeNode(StampNode(metaObj, runTimeUtc), scratch, writer), + { Transformed: true, Meta: { } mutated } => SerializeNode(mutated, scratch, writer), + _ => StampMetadataBytes(er.Metadata, runTimeUtc, scratch, writer) + }; + + // Preserve the ORIGINAL event id (M3) — each event is written once to a fresh dest, and keeping the id + // holds any causation/correlation-by-id references and the idempotency key intact. The type may have + // been renamed; the source content type is preserved. + return new EventData(er.EventId, ctx.Type, dataBytes, metaBytes, er.ContentType); + } + + // Serialise a JsonNode straight to UTF-8 bytes via the pooled writer — no intermediate string. The result + // is copied out (ToArray) because the batch outlives the reused buffer. + private static byte[] SerializeNode(JsonNode node, ArrayBufferWriter scratch, Utf8JsonWriter writer) + { + scratch.ResetWrittenCount(); + writer.Reset(); + node.WriteTo(writer); + writer.Flush(); + return scratch.WrittenSpan.ToArray(); + } + + private static JsonObject StampNode(JsonObject metaObj, DateTime runTimeUtc) + { + metaObj["MigratedAt"] = JsonValue.Create(runTimeUtc); + return metaObj; + } + + // Stamp MigratedAt onto the ORIGINAL metadata bytes without an intermediate string. Empty → fresh object; + // JSON object → copy props (overwriting any existing MigratedAt) + append via the pooled writer; + // non-object JSON or non-JSON → verbatim (a scalar/array cannot carry a property). + private static ReadOnlyMemory StampMetadataBytes(ReadOnlyMemory metaBytes, DateTime runTimeUtc, + ArrayBufferWriter scratch, Utf8JsonWriter writer) + { + if (metaBytes.IsEmpty) + return SerializeNode(new JsonObject { ["MigratedAt"] = JsonValue.Create(runTimeUtc) }, scratch, writer); + + JsonDocument doc; + try { doc = JsonDocument.Parse(metaBytes); } + catch (JsonException) { return metaBytes; } // non-JSON metadata — verbatim + + using (doc) + { + if (doc.RootElement.ValueKind != JsonValueKind.Object) + return metaBytes; // array/scalar metadata — cannot add a property; verbatim + + scratch.ResetWrittenCount(); + writer.Reset(); + writer.WriteStartObject(); + foreach (var prop in doc.RootElement.EnumerateObject()) + { + if (prop.NameEquals("MigratedAt")) continue; // overwrite any pre-existing stamp + prop.WriteTo(writer); + } + writer.WriteString("MigratedAt", runTimeUtc); + writer.WriteEndObject(); + writer.Flush(); + return scratch.WrittenSpan.ToArray(); + } + } + + private static bool IsJsonContent(string? contentType) => + contentType is null || contentType.Contains("json", StringComparison.OrdinalIgnoreCase); + + private static JsonNode? TryParseJson(ReadOnlyMemory bytes) + { + if (bytes.IsEmpty) return null; + try + { + var reader = new Utf8JsonReader(bytes.Span); + return JsonNode.Parse(ref reader); + } + catch (JsonException) + { + return null; + } + } +} diff --git a/src/MicroPlumberd.Migration/IMigrationBuilder.cs b/src/MicroPlumberd.Migration/IMigrationBuilder.cs new file mode 100644 index 0000000..628b81d --- /dev/null +++ b/src/MicroPlumberd.Migration/IMigrationBuilder.cs @@ -0,0 +1,65 @@ +using System.Text.Json.Nodes; + +namespace MicroPlumberd.Migration; + +/// +/// Fluent surface a uses to declare RAW rewrite rules. All operations work on +/// event-type strings and JSON — never on domain event classes. +/// +/// +/// Ordering. Operations are applied to every copied event in the exact order they are +/// declared, and migrations are applied in ascending order. Each operation +/// observes the running result of the operations before it: a declared before a +/// that names the new type will compose, and a +/// declared before a that names the new stream will compose. +/// Short-circuit. Once any drop operation drops an event, later operations for that event are +/// skipped. +/// +public interface IMigrationBuilder +{ + /// Drops (omits from the destination) every event whose current stream equals . + IMigrationBuilder DropStream(string name); + + /// Drops every event whose current stream matches . + IMigrationBuilder DropStream(Func predicate); + + /// Drops every event belonging to any of the given streams. + IMigrationBuilder DropStreams(params string[] names); + + /// Drops individual events matching (the raw view reflects prior operations). + IMigrationBuilder DropEvent(Func predicate); + + /// Rewrites the event-type string from to . + IMigrationBuilder RenameType(string oldType, string newType); + + /// + /// Mutates the JSON payload and metadata in place for every event whose current type equals + /// . Skipped (with a warning) for events whose payload is not JSON. + /// + IMigrationBuilder TransformJson(string type, Action transform); + + /// Retargets every event from stream to stream . + IMigrationBuilder RenameStream(string oldName, string newName); + + /// + /// Registers ONE generic per-event rule: every surviving (non-system, non-link) event is handed to + /// as an immutable , and the returned event replaces it. + /// Returning null — or an event with an empty or + /// — DROPS the event. + /// + /// + /// This is the escape hatch the type-specific operations above are sugar for, and what a + /// script-defined migration compiles to. Like every other operation it is applied in DECLARATION ORDER, so + /// a declared before it short-circuits and the transform never sees those + /// events, while a declared before it is already reflected in + /// . + /// Return the SAME / node instances when + /// you did not change them (e.g. return the argument itself). Change is detected by reference: an + /// untouched payload is then copied BYTE-VERBATIM instead of being re-serialised, which is what keeps a pure + /// copy identical to the source. + /// , and + /// on the returned event are IGNORED: destination streams are renumbered + /// gaplessly and the source event id and timestamp are preserved. + /// + IMigrationBuilder Transform(Func transform); +} diff --git a/src/MicroPlumberd.Migration/MicroPlumberd.Migration.csproj b/src/MicroPlumberd.Migration/MicroPlumberd.Migration.csproj new file mode 100644 index 0000000..5d3baa9 --- /dev/null +++ b/src/MicroPlumberd.Migration/MicroPlumberd.Migration.csproj @@ -0,0 +1,47 @@ + + + + net10.0 + enable + enable + embedded + true + Offline event-store migration / rewrite tool for KurrentDB (EventStore). Copies a source store to a fresh destination in $all commit order, applying raw (non-typed) migration rules — drop streams, drop events, rename event types, transform JSON payload/metadata, rename streams — with applied-migration history that travels with the store. + MicroPlumberd.Migration + Rafal Maciag + Rafal Maciag + logo-squere.png + README.md + MIT + https://github.com/modelingevolution/micro-plumberd + https://modelingevolution.github.io/micro-plumberd/ + EventStore;KurrentDB;CQRS;EventSourcing;Migration + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/src/MicroPlumberd.Migration/Migration.cs b/src/MicroPlumberd.Migration/Migration.cs new file mode 100644 index 0000000..6b4a267 --- /dev/null +++ b/src/MicroPlumberd.Migration/Migration.cs @@ -0,0 +1,42 @@ +namespace MicroPlumberd.Migration; + +/// +/// A single, ordered, RAW event-store migration. A migration has a stable and declares +/// its rewrite rules in . There is intentionally no Up/Down — the copy +/// engine rewrites SOURCE into a fresh DEST, so a migration is forward-only. +/// +/// +/// must be unique and sortable — use a zero-padded numeric prefix, e.g. +/// 0001_drop_tombstoned_test_offers. Migrations run in ascending ordinal order of . +/// Implementations must have a public parameterless constructor so they can be discovered and instantiated. +/// +public abstract class Migration +{ + /// Stable, unique, sortable identifier (e.g. 0001_drop_tombstoned_test_offers). + public abstract string Id { get; } + + /// + /// Optional human-readable name recorded in history. Defaults to . + /// + public virtual string Name => Id; + + /// + /// Optional checksum that REPLACES the default one computed over this migration's compiled operation + /// descriptors. null (the default) keeps the descriptor checksum, which is right for every + /// hand-written migration. + /// + /// + /// Override it only when the rules are defined by an EXTERNAL ARTIFACT rather than by this class's + /// code — a rewrite script, for instance. There, the descriptor checksum is not merely weaker, it is + /// WRONG: every script compiles to the same host delegates, so two completely different scripts produce + /// identical descriptors (same method identity, same IL) and therefore the same checksum. The history + /// guard would then accept a store rewritten by a different script as "already applied". The artifact's + /// own hash — sha256(script text) — is the only thing that identifies those rules. + /// The value is recorded verbatim in and is what the + /// re-run guard compares, so it must change whenever the rules change. + /// + public virtual string? ChecksumOverride => null; + + /// Declares this migration's rewrite rules against . + public abstract void Migrate(IMigrationBuilder b); +} diff --git a/src/MicroPlumberd.Migration/MigrationApplied.cs b/src/MicroPlumberd.Migration/MigrationApplied.cs new file mode 100644 index 0000000..dfa751e --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationApplied.cs @@ -0,0 +1,44 @@ +namespace MicroPlumberd.Migration; + +/// +/// The history record written to the migration-history stream once a migration has been applied to the +/// destination store. One record per applied migration; the history travels with the store. +/// +public sealed record MigrationApplied +{ + /// The migration's . + public required string Id { get; init; } + + /// The migration's human-readable name. + public required string Name { get; init; } + + /// Checksum of the migration's rules at the time it was applied (see ). + public required string Checksum { get; init; } + + /// When the migration run that applied this migration started (UTC). + public required DateTime AppliedAtUtc { get; init; } + + /// Total user events scanned from the source in the run that applied this migration. + public required long SourceEvents { get; init; } + + /// Total user events written to the destination in that run. + public required long Kept { get; init; } + + /// Events this migration dropped. + public required long Dropped { get; init; } + + /// Events this migration transformed (JSON transforms + type/stream renames). + public required long Transformed { get; init; } + + /// + /// The descriptors of the operations this migration compiled to, in declaration order (e.g. + /// DropStreamPredicate(…), RenameType("A"->"B"), Transform(…)) — so an operator + /// reading the history of a store can see WHAT a past rewrite did, not only that it happened. + /// + /// + /// OPTIONAL by contract (RUNBOOK m10): history written by an older tool version has no such field, and a + /// required member here would make those records undeserializable. null means "not recorded", + /// which is not the same as "no operations". + /// + public IReadOnlyList? Descriptors { get; init; } +} diff --git a/src/MicroPlumberd.Migration/MigrationBuilder.cs b/src/MicroPlumberd.Migration/MigrationBuilder.cs new file mode 100644 index 0000000..5171b85 --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationBuilder.cs @@ -0,0 +1,116 @@ +using System.Security.Cryptography; +using System.Text; +using System.Text.Json.Nodes; + +namespace MicroPlumberd.Migration; + +/// Records the ordered operations a single declares. +internal sealed class MigrationBuilder : IMigrationBuilder +{ + private readonly List _ops = new(); + public IReadOnlyList Operations => _ops; + + public IMigrationBuilder DropStream(string name) + { + ArgumentException.ThrowIfNullOrEmpty(name); + _ops.Add(new DropStreamNameOp(name)); + return this; + } + + public IMigrationBuilder DropStream(Func predicate) + { + ArgumentNullException.ThrowIfNull(predicate); + _ops.Add(new DropStreamPredicateOp(predicate)); + return this; + } + + public IMigrationBuilder DropStreams(params string[] names) + { + ArgumentNullException.ThrowIfNull(names); + _ops.Add(new DropStreamsOp(names)); + return this; + } + + public IMigrationBuilder DropEvent(Func predicate) + { + ArgumentNullException.ThrowIfNull(predicate); + _ops.Add(new DropEventOp(predicate)); + return this; + } + + public IMigrationBuilder RenameType(string oldType, string newType) + { + ArgumentException.ThrowIfNullOrEmpty(oldType); + ArgumentException.ThrowIfNullOrEmpty(newType); + _ops.Add(new RenameTypeOp(oldType, newType)); + return this; + } + + public IMigrationBuilder TransformJson(string type, Action transform) + { + ArgumentException.ThrowIfNullOrEmpty(type); + ArgumentNullException.ThrowIfNull(transform); + _ops.Add(new TransformJsonOp(type, transform)); + return this; + } + + public IMigrationBuilder Transform(Func transform) + { + ArgumentNullException.ThrowIfNull(transform); + _ops.Add(new TransformOp(transform)); + return this; + } + + public IMigrationBuilder RenameStream(string oldName, string newName) + { + ArgumentException.ThrowIfNullOrEmpty(oldName); + ArgumentException.ThrowIfNullOrEmpty(newName); + _ops.Add(new RenameStreamOp(oldName, newName)); + return this; + } +} + +/// +/// A whose has been run to capture its ordered +/// operation list, together with a checksum over that list. +/// +internal sealed class CompiledMigration +{ + public required string Id { get; init; } + public required string Name { get; init; } + public required IReadOnlyList Operations { get; init; } + public required string Checksum { get; init; } + + public static CompiledMigration Compile(Migration migration) + { + var b = new MigrationBuilder(); + migration.Migrate(b); + return new CompiledMigration + { + Id = migration.Id, + Name = migration.Name, + Operations = b.Operations, + // A migration whose rules come from an EXTERNAL artifact (a script file) supplies its own checksum + // over that artifact — see Migration.ChecksumOverride for why the descriptor checksum cannot serve. + Checksum = migration.ChecksumOverride ?? ComputeChecksum(migration.Id, b.Operations) + }; + } + + /// + /// SHA-256 over the migration Id and each operation's descriptor. The descriptor captures literal + /// arguments and, for delegate-backed operations, the delegate's method identity and IL body hash — + /// so editing an already-applied migration's rules changes the checksum and the run is refused. + /// + /// + /// Limitation: a pure lambda-body edit is only detected via IL when the method body is reflectable; + /// literal/structural edits are always detected. + /// + internal static string ComputeChecksum(string id, IReadOnlyList ops) + { + var sb = new StringBuilder(); + sb.Append("id=").Append(id).Append('\n'); + for (var i = 0; i < ops.Count; i++) + sb.Append(i).Append(':').Append(ops[i].Descriptor).Append('\n'); + return Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(sb.ToString()))); + } +} diff --git a/src/MicroPlumberd.Migration/MigrationDiscovery.cs b/src/MicroPlumberd.Migration/MigrationDiscovery.cs new file mode 100644 index 0000000..9037ea5 --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationDiscovery.cs @@ -0,0 +1,37 @@ +using System.Reflection; + +namespace MicroPlumberd.Migration; + +/// Discovers and orders implementations. +public static class MigrationDiscovery +{ + /// + /// Finds every concrete with a public parameterless constructor in the given + /// assemblies, instantiates them, and returns them ordered by (ordinal). + /// + public static IReadOnlyList FromAssemblies(params Assembly[] assemblies) + { + ArgumentNullException.ThrowIfNull(assemblies); + var migrations = assemblies + .SelectMany(a => a.GetTypes()) + .Where(t => typeof(Migration).IsAssignableFrom(t) && t is { IsAbstract: false, IsClass: true }) + .Select(t => t.GetConstructor(Type.EmptyTypes) is not null + ? (Migration)Activator.CreateInstance(t)! + : throw new InvalidOperationException( + $"Migration '{t.FullName}' must have a public parameterless constructor.")); + + return Order(migrations); + } + + /// + /// Orders migrations by (ordinal) and rejects duplicate ids. + /// + public static IReadOnlyList Order(IEnumerable migrations) + { + ArgumentNullException.ThrowIfNull(migrations); + var ordered = migrations.OrderBy(m => m.Id, StringComparer.Ordinal).ToList(); + var duplicate = ordered.GroupBy(m => m.Id).FirstOrDefault(g => g.Count() > 1); + if (duplicate is not null) throw new DuplicateMigrationIdException(duplicate.Key); + return ordered; + } +} diff --git a/src/MicroPlumberd.Migration/MigrationExceptions.cs b/src/MicroPlumberd.Migration/MigrationExceptions.cs new file mode 100644 index 0000000..143ed8f --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationExceptions.cs @@ -0,0 +1,28 @@ +namespace MicroPlumberd.Migration; + +/// +/// Thrown when an already-applied migration's code (checksum) differs from what is recorded in the store's +/// history — editing a migration that has already run is refused, because the destination cannot be +/// reproduced from a changed rule set. +/// +public sealed class MigrationChecksumMismatchException(string id, string recorded, string current) + : Exception($"Migration '{id}' was already applied with checksum {recorded}, but its code now hashes to " + + $"{current}. Applied migrations are immutable — add a NEW migration instead of editing this one.") +{ + /// The offending migration's Id. + public string MigrationId { get; } = id; + + /// The checksum recorded in the store's history. + public string RecordedChecksum { get; } = recorded; + + /// The checksum the migration's current code hashes to. + public string CurrentChecksum { get; } = current; +} + +/// Thrown when two defined migrations share the same . +public sealed class DuplicateMigrationIdException(string id) + : Exception($"More than one migration declares Id '{id}'. Migration Ids must be unique.") +{ + /// The duplicated migration Id. + public string MigrationId { get; } = id; +} diff --git a/src/MicroPlumberd.Migration/MigrationHistory.cs b/src/MicroPlumberd.Migration/MigrationHistory.cs new file mode 100644 index 0000000..1efbb6c --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationHistory.cs @@ -0,0 +1,86 @@ +using System.Text; +using System.Text.Json; +using System.Text.Json.Nodes; +using KurrentDB.Client; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MicroPlumberd.Migration; + +/// +/// Reads and writes the applied-migration history that travels inside the store. +/// +/// +/// The history lives in a dedicated stream (). The copy engine treats it as a +/// reserved stream and skips it, so history is never copied twice: this class owns reading it from the +/// source and re-writing it (existing + newly-applied records) to the destination. A non-$ name is +/// used deliberately so writing history needs no $admins privilege — any authenticated connection +/// can author it. (The owner's brief suggested $mp-migrations or a clearly-named non-$ce +/// stream; this is the latter.) +/// +internal sealed class MigrationHistory(ILogger? logger = null) +{ + /// The dedicated history stream. Skipped by the copy engine and reserved from user rules. + public const string StreamName = "mp-migrations"; + + /// Event-type string for a record. + public const string EventType = nameof(MigrationApplied); + + private readonly ILogger _logger = logger ?? NullLogger.Instance; + + private static readonly JsonSerializerOptions JsonOptions = new(JsonSerializerDefaults.General); + + /// Reads all applied-migration records from 's history stream, in order. + public async Task> ReadAsync(KurrentDBClient client, CancellationToken ct = default) + { + var result = client.ReadStreamAsync(Direction.Forwards, StreamName, StreamPosition.Start, + resolveLinkTos: false, cancellationToken: ct); + + if (await result.ReadState.ConfigureAwait(false) == ReadState.StreamNotFound) + { + _logger.LogInformation("No migration history stream ({Stream}) — treating as empty.", StreamName); + return []; + } + + var applied = new List(); + await foreach (var re in result.ConfigureAwait(false)) + { + var er = re.Event; + if (er is null || er.EventType != EventType) continue; + var record = JsonSerializer.Deserialize(er.Data.Span, JsonOptions); + if (record is not null) applied.Add(record); + } + _logger.LogInformation("Read {Count} applied migration(s) from history.", applied.Count); + return applied; + } + + /// + /// Rewrites the full history to 's destination stream: the pre-existing records + /// (verbatim) followed by the newly-applied records. Idempotent per run — call once after a successful copy. + /// + public async Task WriteAsync(KurrentDBClient client, IReadOnlyList existing, + IReadOnlyList newlyApplied, DateTime runTimeUtc, CancellationToken ct = default) + { + var all = existing.Concat(newlyApplied).ToArray(); + if (all.Length == 0) return; + + var events = all.Select(r => ToEventData(r, runTimeUtc)).ToArray(); + // Fresh destination history: expect NoStream. History is authored only by this tool. + await client.AppendToStreamAsync(StreamName, StreamState.NoStream, events, cancellationToken: ct) + .ConfigureAwait(false); + _logger.LogInformation("Wrote {Total} migration history record(s) to {Stream} ({New} newly applied).", + all.Length, StreamName, newlyApplied.Count); + } + + private static EventData ToEventData(MigrationApplied record, DateTime runTimeUtc) + { + var data = JsonSerializer.SerializeToUtf8Bytes(record, JsonOptions); + var meta = new JsonObject + { + ["Created"] = record.AppliedAtUtc, + ["MigratedAt"] = runTimeUtc + }; + var metaBytes = Encoding.UTF8.GetBytes(meta.ToJsonString()); + return new EventData(Uuid.NewUuid(), EventType, data, metaBytes); + } +} diff --git a/src/MicroPlumberd.Migration/MigrationOps.cs b/src/MicroPlumberd.Migration/MigrationOps.cs new file mode 100644 index 0000000..c412dc2 --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationOps.cs @@ -0,0 +1,287 @@ +using System.Reflection; +using System.Security.Cryptography; +using System.Text; +using System.Text.Json.Nodes; + +namespace MicroPlumberd.Migration; + +/// The effect a single operation had on one event. +internal enum OpEffect +{ + None, + Dropped, + Renamed, + Transformed +} + +/// +/// Mutable running state of one event as it is folded through the ordered operation list. +/// +internal sealed class EventContext +{ + // All settable so the copy loop can REUSE one instance per event (reset each iteration) — it never escapes + // except into the synchronous DropEvent predicate via the immutable AsRawEvent() snapshot. + public required string SourceStream { get; set; } + public required ulong EventNumber { get; set; } + public required string TargetStream { get; set; } + public required string Type { get; set; } + public JsonNode? Data { get; set; } + public JsonNode? Meta { get; set; } + public bool Dropped { get; set; } + + /// The SOURCE event id (never rewritten — the copy engine preserves it on the destination). + public Guid EventId { get; set; } + + /// The SOURCE write timestamp (UTC), exposed to rules so they can match by date. + public DateTime Created { get; set; } + + /// + /// True once a TransformJson rule mutated this event's payload/metadata — the copy engine then + /// re-serialises /. When false the original bytes are copied VERBATIM + /// (no parse, no re-serialise), even if the stream/type was renamed (those don't touch the payload). + /// + public bool Transformed { get; set; } + + /// Builds an immutable view reflecting the current running state. + public RawEvent AsRawEvent() => new() + { + StreamId = TargetStream, + EventNumber = EventNumber, + Type = Type, + Data = Data, + Metadata = Meta, + EventId = EventId, + Created = Created + }; +} + +/// A single compiled migration operation. +internal interface IMigrationOp +{ + /// Deterministic textual description used for checksumming a migration's rules. + string Descriptor { get; } + + /// Applies the operation to , returning the effect it had. + OpEffect Apply(EventContext ctx, IMigrationOpLog? log); + + /// Static stream names this op references (for reserved-name validation). Empty when dynamic. + IEnumerable ReferencedStreamNames { get; } +} + +/// Optional sink for per-op warnings during evaluation (e.g. transform skipped on non-JSON). +internal interface IMigrationOpLog +{ + void TransformSkippedNonJson(string type, string stream, ulong eventNumber); +} + +internal static class DelegateChecksum +{ + /// + /// Produces a checksum fragment for a delegate that is sensitive to BOTH which method backs it and + /// the method's IL body — so editing a lambda's body changes the checksum. Falls back to the method + /// identity when IL is unavailable (e.g. runtime-generated methods). + /// + public static string Describe(Delegate d) + { + var m = d.Method; + var sb = new StringBuilder(); + sb.Append(m.DeclaringType?.FullName).Append('.').Append(m.Name); + try + { + var il = m.GetMethodBody()?.GetILAsByteArray(); + if (il is { Length: > 0 }) + { + var hash = SHA256.HashData(il); + sb.Append(":il=").Append(Convert.ToHexString(hash)); + } + } + catch + { + // IL not accessible — method identity above is the best we can do. + } + return sb.ToString(); + } +} + +internal sealed class DropStreamNameOp(string name) : IMigrationOp +{ + public string Descriptor => $"DropStream(\"{name}\")"; + public IEnumerable ReferencedStreamNames => [name]; + + public OpEffect Apply(EventContext ctx, IMigrationOpLog? log) + { + if (ctx.Dropped || ctx.TargetStream != name) return OpEffect.None; + ctx.Dropped = true; + return OpEffect.Dropped; + } +} + +internal sealed class DropStreamsOp(string[] names) : IMigrationOp +{ + private readonly HashSet _names = new(names, StringComparer.Ordinal); + public string Descriptor => $"DropStreams({string.Join(",", names.Select(n => $"\"{n}\""))})"; + public IEnumerable ReferencedStreamNames => names; + + public OpEffect Apply(EventContext ctx, IMigrationOpLog? log) + { + if (ctx.Dropped || !_names.Contains(ctx.TargetStream)) return OpEffect.None; + ctx.Dropped = true; + return OpEffect.Dropped; + } +} + +internal sealed class DropStreamPredicateOp(Func predicate) : IMigrationOp +{ + public string Descriptor => $"DropStreamPredicate({DelegateChecksum.Describe(predicate)})"; + public IEnumerable ReferencedStreamNames => []; + + public OpEffect Apply(EventContext ctx, IMigrationOpLog? log) + { + if (ctx.Dropped || !predicate(ctx.TargetStream)) return OpEffect.None; + ctx.Dropped = true; + return OpEffect.Dropped; + } +} + +internal sealed class DropEventOp(Func predicate) : IMigrationOp +{ + public string Descriptor => $"DropEvent({DelegateChecksum.Describe(predicate)})"; + public IEnumerable ReferencedStreamNames => []; + + public OpEffect Apply(EventContext ctx, IMigrationOpLog? log) + { + if (ctx.Dropped || !predicate(ctx.AsRawEvent())) return OpEffect.None; + ctx.Dropped = true; + return OpEffect.Dropped; + } +} + +internal sealed class RenameTypeOp(string oldType, string newType) : IMigrationOp +{ + /// The new event-type name this rule assigns (validated against $-space by the runner). + public string NewType => newType; + + public string Descriptor => $"RenameType(\"{oldType}\"->\"{newType}\")"; + public IEnumerable ReferencedStreamNames => []; + + public OpEffect Apply(EventContext ctx, IMigrationOpLog? log) + { + if (ctx.Dropped || ctx.Type != oldType) return OpEffect.None; + ctx.Type = newType; + return OpEffect.Renamed; + } +} + +internal sealed class TransformJsonOp(string type, Action transform) : IMigrationOp +{ + /// The event type this transform targets (used by the plan to decide which types need parsing). + public string Type => type; + + public string Descriptor => $"TransformJson(\"{type}\",{DelegateChecksum.Describe(transform)})"; + public IEnumerable ReferencedStreamNames => []; + + public OpEffect Apply(EventContext ctx, IMigrationOpLog? log) + { + if (ctx.Dropped || ctx.Type != type) return OpEffect.None; + if (ctx.Data is null) + { + // Non-JSON payload: cannot hand the caller a JsonNode. Skip rather than fabricate data. + log?.TransformSkippedNonJson(type, ctx.TargetStream, ctx.EventNumber); + return OpEffect.None; + } + ctx.Meta ??= new JsonObject(); + transform(ctx.Data, ctx.Meta); + ctx.Transformed = true; // payload/metadata mutated → the copy engine must re-serialise, not copy verbatim + return OpEffect.Transformed; + } +} + +internal sealed class RenameStreamOp(string oldName, string newName) : IMigrationOp +{ + /// The new stream name this rule retargets events to (validated against $-space/reserved by the runner). + public string NewName => newName; + + public string Descriptor => $"RenameStream(\"{oldName}\"->\"{newName}\")"; + public IEnumerable ReferencedStreamNames => [oldName, newName]; + + public OpEffect Apply(EventContext ctx, IMigrationOpLog? log) + { + if (ctx.Dropped || ctx.TargetStream != oldName) return OpEffect.None; + ctx.TargetStream = newName; + return OpEffect.Renamed; + } +} + +/// +/// The one GENERIC per-event operation: hands each surviving event to as an +/// immutable and folds the returned event back into the running state. Returning +/// null DROPS the event. +/// +/// +/// Unlike (which targets ONE event type and mutates the payload in place), +/// this op sees EVERY non-system, non-link event and may change the target stream, the event type, the payload +/// and the metadata in one pass. It is what a script-defined migration compiles to. +/// Change detection is by REFERENCE. The returned / +/// are compared to the ones handed in with ReferenceEquals: a transform that did not touch the +/// payload returns the SAME node, the event is therefore NOT marked transformed, and the copy engine writes the +/// ORIGINAL BYTES verbatim. That is what keeps a pure copy byte-identical — re-serialising an untouched payload +/// would silently re-render its JSON (number formatting, key escapes) for every event in the store. +/// , and +/// on the returned event are IGNORED — the copy engine renumbers each destination stream gaplessly and +/// preserves the source event id and write timestamp. +/// +internal sealed class TransformOp(Func transform) : IMigrationOp +{ + public string Descriptor => $"Transform({DelegateChecksum.Describe(transform)})"; + public IEnumerable ReferencedStreamNames => []; + + public OpEffect Apply(EventContext ctx, IMigrationOpLog? log) + { + if (ctx.Dropped) return OpEffect.None; + + var result = transform(ctx.AsRawEvent()); + if (result is null) + { + ctx.Dropped = true; + return OpEffect.Dropped; + } + + // An empty stream or event type is the script contract's second way of saying "drop" (Replicator). + if (result.StreamId.Length == 0 || result.Type.Length == 0) + { + ctx.Dropped = true; + return OpEffect.Dropped; + } + + var renamed = false; + if (!string.Equals(result.StreamId, ctx.TargetStream, StringComparison.Ordinal)) + { + ctx.TargetStream = result.StreamId; + renamed = true; + } + if (!string.Equals(result.Type, ctx.Type, StringComparison.Ordinal)) + { + ctx.Type = result.Type; + renamed = true; + } + + var transformed = false; + if (!ReferenceEquals(result.Data, ctx.Data)) + { + ctx.Data = result.Data; + transformed = true; + } + if (!ReferenceEquals(result.Metadata, ctx.Meta)) + { + ctx.Meta = result.Metadata; + transformed = true; + } + + if (transformed) + { + ctx.Transformed = true; // payload/metadata changed → the copy engine must re-serialise + return OpEffect.Transformed; + } + return renamed ? OpEffect.Renamed : OpEffect.None; + } +} diff --git a/src/MicroPlumberd.Migration/MigrationPlan.cs b/src/MicroPlumberd.Migration/MigrationPlan.cs new file mode 100644 index 0000000..ebd49f7 --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationPlan.cs @@ -0,0 +1,119 @@ +namespace MicroPlumberd.Migration; + +/// Per-migration effect counters accumulated over a run (or a dry run). +public sealed class MigrationStats +{ + /// The migration's Id. + public required string Id { get; init; } + + /// The migration's name. + public required string Name { get; init; } + + /// Events this migration dropped. + public long Dropped { get; internal set; } + + /// Type/stream renames this migration performed. + public long Renamed { get; internal set; } + + /// JSON transforms this migration performed. + public long Transformed { get; internal set; } +} + +/// +/// The composed set of PENDING migrations, applied in order to each event. Owns per-migration attribution. +/// +internal sealed class MigrationPlan +{ + private readonly (string MigId, IMigrationOp Op)[] _ops; + + public IReadOnlyList Migrations { get; } + public IReadOnlyDictionary Stats { get; } + + /// Event types that a TransformJson rule targets — the only types whose payload needs parsing. + public IReadOnlySet TransformTypes { get; } + + /// + /// True if any DropEvent(Func<RawEvent,bool>) rule exists. exposes the + /// payload/metadata as PARSED , and the predicate is opaque, so + /// any such rule conservatively forces every event to be parsed (we cannot know which fields it reads). + /// + public bool AnyDropEventRule { get; } + + /// + /// True if any generic Transform(Func<RawEvent,RawEvent?>) rule exists. Such a rule is handed + /// EVERY event as a whose payload/metadata are PARSED, and the delegate is opaque, + /// so — exactly as for — every event must be parsed. + /// + public bool AnyTransformRule { get; } + + public MigrationPlan(IReadOnlyList migrations) + { + Migrations = migrations; + _ops = migrations.SelectMany(m => m.Operations.Select(op => (m.Id, op))).ToArray(); + Stats = migrations.ToDictionary( + m => m.Id, + m => new MigrationStats { Id = m.Id, Name = m.Name }); + + var transformTypes = new HashSet(StringComparer.Ordinal); + var anyDropEvent = false; + var anyTransform = false; + foreach (var (_, op) in _ops) + { + switch (op) + { + case TransformJsonOp t: transformTypes.Add(t.Type); break; + case DropEventOp: anyDropEvent = true; break; + case TransformOp: anyTransform = true; break; + } + } + TransformTypes = transformTypes; + AnyDropEventRule = anyDropEvent; + AnyTransformRule = anyTransform; + } + + /// + /// Whether the copy engine must parse an event of to apply the rules. False for + /// the common case (no rule inspects/transforms the payload) → the event is copied byte-verbatim. + /// + public bool RequiresPayloadParse(string eventType) => + AnyDropEventRule || AnyTransformRule || TransformTypes.Contains(eventType); + + /// Folds through every operation, updating per-migration . + public void Apply(EventContext ctx, IMigrationOpLog? log) + { + foreach (var (migId, op) in _ops) + { + var effect = op.Apply(ctx, log); + if (effect == OpEffect.None) continue; + + var s = Stats[migId]; + switch (effect) + { + case OpEffect.Dropped: s.Dropped++; break; + case OpEffect.Renamed: s.Renamed++; break; + case OpEffect.Transformed: s.Transformed++; break; + } + if (ctx.Dropped) break; // nothing further can affect a dropped event + } + } + + /// + /// Static stream names any operation references — validated against reserved streams before a run so a + /// rule can never target the migration-history stream. + /// + public IEnumerable ReferencedStreamNames => + _ops.SelectMany(t => t.Op.ReferencedStreamNames); + + /// Every operation descriptor of every pending migration, keyed by migration id, in order. + public IReadOnlyList DescriptorsOf(string migrationId) => + _ops.Where(t => string.Equals(t.MigId, migrationId, StringComparison.Ordinal)) + .Select(t => t.Op.Descriptor).ToList(); + + /// New stream names that a RenameStream rule retargets events into (validated by the runner). + public IEnumerable RenameStreamTargets => + _ops.Select(t => t.Op).OfType().Select(o => o.NewName); + + /// New event-type names that a RenameType rule assigns (validated by the runner). + public IEnumerable RenameTypeTargets => + _ops.Select(t => t.Op).OfType().Select(o => o.NewType); +} diff --git a/src/MicroPlumberd.Migration/MigrationProgress.cs b/src/MicroPlumberd.Migration/MigrationProgress.cs new file mode 100644 index 0000000..65cdc67 --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationProgress.cs @@ -0,0 +1,138 @@ +using System.Collections.Immutable; + +namespace MicroPlumberd.Migration; + +/// Overall state of a migration run. +public enum MigrationRunState +{ + /// The run has not started yet. + Pending, + + /// The run is in progress. + Running, + + /// The run finished successfully. + Completed, + + /// The run threw and did not finish. + Failed +} + +/// The current phase of a running migration. +public enum MigrationPhase +{ + /// Nothing has started. + NotStarted, + + /// Reading applied history and computing the pending set. + Planning, + + /// Pre-creating the app's join projections on the empty dest (before the event copy). + ProjectionCopy, + + /// Copying source events to the dest in $all order (paced so the projections stay caught up). + EventCopy, + + /// Final drain of the join projections to head. + Draining, + + /// Appending the migration-history records to the dest. + RecordingHistory, + + /// Cross-checking the dest against the copy bookkeeping. + Verifying, + + /// The run finished successfully. + Completed, + + /// The run failed. + Failed +} + +/// Lifecycle state of a single migration within a run. +public enum MigrationItemState +{ + /// Defined but not yet applied (or applied in a previous run and carried forward — see below). + Pending, + + /// Being applied in the current run (pending migrations are applied together in one $all pass). + Running, + + /// Applied — either recorded in this run or already present in the source history. + Completed, + + /// The run failed while this migration was being applied. + Failed +} + +/// Per-migration status. Serializable; safe to hand to a REST layer. +/// The migration id (ordering key). +/// The migration's human name. +/// Its lifecycle state in this run. +/// When it entered (null if never). +/// When it reached / (null if never). +/// The failure message, if it failed. +public sealed record MigrationItemStatus( + string Id, + string Name, + MigrationItemState State, + DateTime? StartedUtc, + DateTime? FinishedUtc, + string? Error); + +/// +/// Progress of the event-copy phase. Progress is measured by $all commit position (the position of the +/// last COPYABLE event is captured up front) rather than a pre-counted total — counting every source event +/// would require a second full $all scan of the whole store. The denominator excludes the trailing +/// $-system / $> merge-link events the copy skips, so the percentage reaches 100 when the last +/// copyable event is processed. is the running count of user events written so far. +/// +/// User events written to the dest so far. +/// The source stream of the event most recently processed. +/// The $all commit position most recently processed. +/// The $all commit position of the last copyable source event, captured before the copy. +public sealed record EventCopyProgress( + long EventsCopied, + string? CurrentSourceStream, + ulong CurrentCommitPosition, + ulong HeadCommitPosition) +{ + /// 0–100, from commit position vs the head captured up front. 100 when the head is 0 (empty source). + public double PercentComplete => HeadCommitPosition == 0 + ? 100d + : Math.Min(100d, CurrentCommitPosition * 100d / HeadCommitPosition); +} + +/// +/// An immutable, serializable snapshot of a migration run's status + per-phase progress. Safe to read +/// concurrently while the run proceeds (each read returns a fresh immutable instance). +/// +/// Overall run state. +/// The current phase. +/// Whether this is a dry run (nothing is written). +/// +/// Representative id of the pending migrations being applied while running (the tool applies all pending +/// migrations together in a single $all pass, so this is the first pending id; null when not running or +/// nothing is pending). The authoritative per-migration view is . +/// +/// Every migration (carried-forward + pending) with its per-item state. +/// Event-copy progress (null before the copy starts). +/// When the run started. +/// When the run finished (null while running). +/// The run's failure message, if it failed. +public sealed record MigrationStatusSnapshot( + MigrationRunState RunState, + MigrationPhase Phase, + bool DryRun, + string? RunningMigrationId, + ImmutableArray Migrations, + EventCopyProgress? EventCopy, + DateTime? StartedUtc, + DateTime? FinishedUtc, + string? Error) +{ + /// The initial "nothing has happened yet" snapshot. + public static readonly MigrationStatusSnapshot Empty = new( + MigrationRunState.Pending, MigrationPhase.NotStarted, false, null, + ImmutableArray.Empty, null, null, null, null); +} diff --git a/src/MicroPlumberd.Migration/MigrationProgressTracker.cs b/src/MicroPlumberd.Migration/MigrationProgressTracker.cs new file mode 100644 index 0000000..6593026 --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationProgressTracker.cs @@ -0,0 +1,238 @@ +using System.Collections.Immutable; + +namespace MicroPlumberd.Migration; + +/// +/// The migration run's live status + progress. A single instance is owned by a +/// (see ); the runner and its copy/drain phases push updates into it and +/// consumers pull (e.g. a REST endpoint serializes it) or subscribe via +/// / (e.g. a Blazor page updates live). +/// +/// +/// Thread-safe: all mutable state is guarded by a single lock and every read returns a fresh immutable +/// , so is safe to read concurrently while the +/// run proceeds. The chatty per-event updates are coalesced: internal counters advance on every event, but +/// /observers are notified at most once per (plus +/// always on phase/state transitions), so a million-event copy does not raise a million notifications. +/// +public sealed class MigrationProgressTracker : IObservable +{ + /// Minimum wall-clock gap between throttled progress notifications during the event copy. + public static readonly TimeSpan PublishThrottle = TimeSpan.FromMilliseconds(250); + + private readonly object _gate = new(); + private readonly List> _observers = new(); + + // Mutable run state (all access under _gate). + private MigrationRunState _runState = MigrationRunState.Pending; + private MigrationPhase _phase = MigrationPhase.NotStarted; + private bool _dryRun; + private string? _runningMigrationId; + private readonly List _items = new(); + // Event-copy progress is held as flat mutable fields (NOT a record) so the hot per-event report + // allocates NOTHING — the EventCopyProgress record is materialised only when a snapshot is built. + private bool _eventCopyStarted; + private long _eventsCopied; + private string? _eventCurrentStream; + private ulong _eventCurrentPos; + private ulong _eventHeadPos; + private DateTime? _startedUtc; + private DateTime? _finishedUtc; + private string? _error; + private static readonly long ThrottleMs = (long)PublishThrottle.TotalMilliseconds; + private long _lastPublishTicks = long.MinValue; + + private sealed class MutableItem + { + public required string Id; + public required string Name; + public MigrationItemState State; + public DateTime? StartedUtc; + public DateTime? FinishedUtc; + public string? Error; + } + + /// Raised whenever the status changes (throttled during the event copy). The argument is a fresh snapshot. + public event EventHandler? ProgressChanged; + + /// A thread-safe, immutable snapshot of the run's current status and per-phase progress. + public MigrationStatusSnapshot CurrentStatus + { + get { lock (_gate) return BuildSnapshotLocked(); } + } + + /// Rx subscription: the observer receives the current snapshot immediately, then every change. + public IDisposable Subscribe(IObserver observer) + { + MigrationStatusSnapshot snap; + lock (_gate) + { + _observers.Add(observer); + snap = BuildSnapshotLocked(); + } + observer.OnNext(snap); + return new Unsubscriber(this, observer); + } + + // ---- internal reporting API (called by the runner + copy/drain phases) ------------------------------- + + internal void SetMigrations(IEnumerable<(string Id, string Name)> applied, + IEnumerable<(string Id, string Name)> pending) + { + lock (_gate) + { + _items.Clear(); + foreach (var (id, name) in applied) + _items.Add(new MutableItem { Id = id, Name = name, State = MigrationItemState.Completed }); + foreach (var (id, name) in pending) + _items.Add(new MutableItem { Id = id, Name = name, State = MigrationItemState.Pending }); + } + Publish(force: true); + } + + internal void BeginRun(bool dryRun, DateTime startedUtc) + { + lock (_gate) + { + _dryRun = dryRun; + _runState = MigrationRunState.Running; + _startedUtc = startedUtc; + _phase = MigrationPhase.Planning; + } + Publish(force: true); + } + + internal void EnterPhase(MigrationPhase phase) + { + lock (_gate) _phase = phase; + Publish(force: true); + } + + /// Marks all pending migrations Running (they are applied together in the single copy pass). + internal void MarkPendingRunning(DateTime startedUtc) + { + lock (_gate) + { + string? first = null; + foreach (var i in _items.Where(i => i.State == MigrationItemState.Pending)) + { + i.State = MigrationItemState.Running; + i.StartedUtc = startedUtc; + first ??= i.Id; + } + _runningMigrationId = first; + } + Publish(force: true); + } + + internal void MarkPendingCompleted(DateTime finishedUtc) + { + lock (_gate) + { + foreach (var i in _items.Where(i => i.State == MigrationItemState.Running)) + { + i.State = MigrationItemState.Completed; + i.FinishedUtc = finishedUtc; + } + _runningMigrationId = null; + } + Publish(force: true); + } + + internal void EventCopyPlan(ulong headCommitPosition) + { + lock (_gate) + { + _eventCopyStarted = true; + _eventHeadPos = headCommitPosition; + _eventsCopied = 0; + _eventCurrentStream = null; + _eventCurrentPos = 0; + } + Publish(force: true); + } + + // HOT PATH — called once per source event. Updates flat fields only (no record/LINQ/closure/boxing + // allocation) and throttles the notification, so a million-event copy costs a lock + a few field writes. + internal void EventCopied(long eventsCopied, string currentSourceStream, ulong currentCommitPosition) + { + lock (_gate) + { + _eventsCopied = eventsCopied; + _eventCurrentStream = currentSourceStream; + _eventCurrentPos = currentCommitPosition; + } + Publish(force: false); // throttled + } + + internal void Complete(DateTime finishedUtc) + { + lock (_gate) + { + _runState = MigrationRunState.Completed; + _phase = MigrationPhase.Completed; + _finishedUtc = finishedUtc; + } + Publish(force: true); + } + + internal void Fail(string error, DateTime finishedUtc) + { + lock (_gate) + { + _runState = MigrationRunState.Failed; + _phase = MigrationPhase.Failed; + _finishedUtc = finishedUtc; + _error = error; + foreach (var i in _items.Where(i => i.State == MigrationItemState.Running)) + { + i.State = MigrationItemState.Failed; + i.FinishedUtc = finishedUtc; + i.Error = error; + } + _runningMigrationId = null; + } + Publish(force: true); + } + + // ---- snapshot + notification ------------------------------------------------------------------------- + + private MigrationStatusSnapshot BuildSnapshotLocked() + { + var items = _items + .Select(i => new MigrationItemStatus(i.Id, i.Name, i.State, i.StartedUtc, i.FinishedUtc, i.Error)) + .ToImmutableArray(); + // Materialise the event-copy record only here (snapshot build), never on the per-event hot path. + var eventCopy = _eventCopyStarted + ? new EventCopyProgress(_eventsCopied, _eventCurrentStream, _eventCurrentPos, _eventHeadPos) + : null; + return new MigrationStatusSnapshot(_runState, _phase, _dryRun, _runningMigrationId, items, + eventCopy, _startedUtc, _finishedUtc, _error); + } + + private void Publish(bool force) + { + MigrationStatusSnapshot snap; + IObserver[] observers; + lock (_gate) + { + // On the throttled (early-return) path nothing is allocated — no snapshot, no observer copy. + var now = Environment.TickCount64; + if (!force && now - _lastPublishTicks < ThrottleMs) return; + _lastPublishTicks = now; + snap = BuildSnapshotLocked(); + observers = _observers.Count == 0 ? Array.Empty>() : _observers.ToArray(); + } + ProgressChanged?.Invoke(this, snap); + foreach (var o in observers) o.OnNext(snap); + } + + private sealed class Unsubscriber(MigrationProgressTracker owner, IObserver observer) + : IDisposable + { + public void Dispose() + { + lock (owner._gate) owner._observers.Remove(observer); + } + } +} diff --git a/src/MicroPlumberd.Migration/MigrationRunner.cs b/src/MicroPlumberd.Migration/MigrationRunner.cs new file mode 100644 index 0000000..d1e7edf --- /dev/null +++ b/src/MicroPlumberd.Migration/MigrationRunner.cs @@ -0,0 +1,288 @@ +using KurrentDB.Client; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MicroPlumberd.Migration; + +/// The outcome of a full migration run (dry or real). +public sealed class MigrationRunResult +{ + /// True when the run wrote nothing to the destination. + public required bool DryRun { get; init; } + + /// Ids of the migrations that were pending and applied (in order). + public required IReadOnlyList PendingMigrationIds { get; init; } + + /// The copy engine's result (per-migration and per-stream accounting). + public required CopyResult Copy { get; init; } + + /// The verification report — null on a dry run (nothing was written). + public VerificationReport? Verification { get; init; } + + /// The history records appended to the destination — empty on a dry run. + public required IReadOnlyList NewlyApplied { get; init; } + + /// Names of the source user projections pre-created on the dest (empty when no projection copy ran). + public IReadOnlyList CopiedProjections { get; init; } = []; + + /// + /// User-defined indexes created on the destination to serve the [OutputStream] merges in commit order + /// (empty unless a ran). REWIRING REQUIRED: the physical + /// merge streams are NOT rebuilt — for each entry a consumer must read + /// ($idx-user-…) INSTEAD of + /// (which is left empty on the dest). See . + /// + public IReadOnlyList CreatedIndexes { get; init; } = []; +} + +/// +/// Enables the PROJECTION COPY (the owner's [OutputStream] recovery design). When supplied to +/// , the runner pre-creates the app's linkTo join projections on +/// the empty destination and PACES the event copy so each projection stays caught up event-by-event (never a +/// backlog), rebuilding every merge stream in commit order; the app NO-OPs on boot via the reproduced +/// mp_query_hash. When omitted, the runner runs the simple copy (merge streams are skipped and the app +/// regenerates them on boot — order-correct only if the app's projection is itself commit-ordered). +/// +public sealed class ProjectionCopyContext +{ + /// Projection-management client for the SOURCE store (read the user projections + their queries). + public required KurrentDBProjectionManagementClient SourceProjections { get; init; } + + /// Projection-management client for the DEST store (create + drain the live projections). + public required KurrentDBProjectionManagementClient DestProjections { get; init; } + + /// SOURCE connection string — used to read each projection's query verbatim over HTTP. + public required string SourceConnectionString { get; init; } + + /// Per-event pacing timeout (max wait for one link to be emitted). Default 30s. + public TimeSpan PerEventPaceTimeout { get; init; } = TimeSpan.FromSeconds(30); + + /// Final per-projection drain timeout. Default 60s. + public TimeSpan DrainTimeout { get; init; } = TimeSpan.FromSeconds(60); +} + +/// +/// Orchestrates an offline migration run: read history, guard checksums, compute the pending set, copy +/// SOURCE→DEST applying the pending migrations, then (for a real run) record history and verify. +/// +public sealed class MigrationRunner(ILoggerFactory? loggerFactory = null) +{ + private readonly ILoggerFactory _lf = loggerFactory ?? NullLoggerFactory.Instance; + + /// + /// Live status + per-phase progress of this runner's current/last . Subscribe to + /// (or the ) for live + /// updates, or read at any time (thread-safe). One + /// runner instance drives one run — hold it in the migrator app and serialize CurrentStatus in REST. + /// + public MigrationProgressTracker Progress { get; } = new(); + + /// + /// The stream the applied-migration history lives in. Reserved: the copy engine never copies it, and no + /// rule may target it. Exposed because a caller that reads the source independently of the engine has to + /// exclude exactly the same stream — deriving that from a second copy of the literal is how the two drift. + /// + public static string HistoryStreamName => MigrationHistory.StreamName; + + /// Shorthand for Progress.CurrentStatus — the thread-safe status snapshot for a REST layer. + public MigrationStatusSnapshot CurrentStatus => Progress.CurrentStatus; + + /// + /// Runs the migration. is read-only; must be a fresh + /// (empty) store. When is true, nothing is written to . + /// Progress is reported through throughout the run. + /// + public async Task RunAsync( + KurrentDBClient source, + KurrentDBClient dest, + IReadOnlyList migrations, + bool dryRun, + ProjectionCopyContext? projectionCopy = null, + UserDefinedIndexCopyContext? indexCopy = null, + CancellationToken ct = default) + { + ArgumentNullException.ThrowIfNull(source); + ArgumentNullException.ThrowIfNull(dest); + ArgumentNullException.ThrowIfNull(migrations); + if (projectionCopy is not null && indexCopy is not null) + throw new InvalidOperationException( + "ProjectionCopyContext and UserDefinedIndexCopyContext are mutually exclusive merge-recovery " + + "strategies — supply at most one. The projection copy PACES a rebuild of the physical merge " + + "streams; the index copy creates user-defined indexes read in commit order."); + + var logger = _lf.CreateLogger(); + var runTimeUtc = DateTime.UtcNow; + var reserved = new HashSet(StringComparer.Ordinal) { MigrationHistory.StreamName }; + + Progress.BeginRun(dryRun, runTimeUtc); + try + { + // 1. Order defined migrations (rejects duplicate Ids) and compile their rules. + var defined = MigrationDiscovery.Order(migrations).Select(CompiledMigration.Compile).ToList(); + + // 2. Read applied history and enforce the checksum guard. + var history = new MigrationHistory(_lf.CreateLogger()); + var applied = await history.ReadAsync(source, ct).ConfigureAwait(false); + var appliedById = applied.ToDictionary(a => a.Id, StringComparer.Ordinal); + foreach (var m in defined) + { + if (appliedById.TryGetValue(m.Id, out var a) && a.Checksum != m.Checksum) + throw new MigrationChecksumMismatchException(m.Id, a.Checksum, m.Checksum); + } + foreach (var a in applied) + if (defined.All(d => d.Id != a.Id)) + logger.LogWarning("History contains applied migration '{Id}' not present in code — its " + + "effects are already baked into the source and it will be carried forward.", a.Id); + + // 3. Pending = defined not yet applied, in Id order. + var pending = defined.Where(d => !appliedById.ContainsKey(d.Id)).ToList(); + var plan = new MigrationPlan(pending); + Progress.SetMigrations( + applied.Select(a => (a.Id, a.Name)), + pending.Select(p => (p.Id, p.Name))); + + // 4. Reserved/system guard (m8a): no rule may target the reserved history stream, and no rule may + // retarget an event into $-space — a RenameStream into $… or a RenameType to a $-typed name would + // be silently SKIPPED by the copy engine (system-namespace), making those events vanish. + static bool IsSystem(string n) => n.Length > 0 && n[0] == '$'; + var offending = plan.ReferencedStreamNames.Concat(plan.RenameStreamTargets) + .Where(n => reserved.Contains(n) || IsSystem(n)).Distinct().ToList(); + var offendingTypes = plan.RenameTypeTargets.Where(IsSystem).Distinct().ToList(); + if (offending.Count > 0 || offendingTypes.Count > 0) + throw new InvalidOperationException( + "Migration rules must not target the reserved history stream or the $-system namespace. " + + $"Offending stream target(s): {(offending.Count == 0 ? "(none)" : string.Join(", ", offending))}; " + + $"offending event-type target(s): {(offendingTypes.Count == 0 ? "(none)" : string.Join(", ", offendingTypes))}."); + + // m8b: migration ids are ordered ORDINALLY; mixed-width numeric prefixes silently misorder + // (e.g. "10" sorts before "2"). Warn so the author zero-pads to equal width. + var prefixWidths = defined.Select(d => LeadingDigitWidth(d.Id)).Where(w => w > 0).Distinct().ToList(); + if (prefixWidths.Count > 1) + logger.LogWarning("Migration ids have INCONSISTENT numeric-prefix widths ({Widths}) — ordinal " + + "ordering may misorder them (e.g. \"10\" before \"2\"). Zero-pad ids to equal width.", + string.Join(", ", prefixWidths.OrderBy(w => w))); + + logger.LogInformation("{Count} pending migration(s): {Ids}", pending.Count, + pending.Count == 0 ? "(none)" : string.Join(", ", pending.Select(p => p.Id))); + + // 4b. PROJECTION COPY (optional): pre-create the app's join projections on the EMPTY dest so they are + // the sole writers of the [OutputStream] merge streams, then PACE the copy below so they stay + // caught up event-by-event (no backlog → fromStreams can't type-cluster). Only $-system streams + // are added here, so the copy engine's dest-empty pre-flight still passes. + var copier = projectionCopy is null ? null : new ProjectionCopier(_lf.CreateLogger()); + IReadOnlyList copied = []; + IEventPacer? pacer = null; + if (projectionCopy is not null) + { + Progress.EnterPhase(MigrationPhase.ProjectionCopy); + await copier!.EnsureStandardProjectionsRunningAsync(projectionCopy.DestProjections, dryRun, ct) + .ConfigureAwait(false); + copied = await copier.DiscoverAsync(projectionCopy.SourceProjections, + projectionCopy.SourceConnectionString, ct).ConfigureAwait(false); + logger.LogInformation("Projection copy: {Count} user projection(s) to {Verb}: {Names}", + copied.Count, dryRun ? "create (dry run — not written)" : "pre-create", + copied.Count == 0 ? "(none)" : string.Join(", ", copied.Select(p => p.Name))); + await copier.PreCreateAsync(copied, projectionCopy.DestProjections, dest, dryRun, ct) + .ConfigureAwait(false); + if (!dryRun && copied.Count > 0) + pacer = copier.CreatePacer(copied, dest, projectionCopy.PerEventPaceTimeout); + } + + // 5. Copy (or dry run) — reads source $all in commit order and writes dest in the same order. Without + // a projection copy the merge/link ($>) streams are SKIPPED and the app regenerates them on boot; + // with one, the pacer keeps the pre-created join projections caught up event-by-event. + Progress.EnterPhase(MigrationPhase.EventCopy); + if (!dryRun) Progress.MarkPendingRunning(runTimeUtc); // all pending applied together in one pass + var copy = await new CopyEngine(_lf.CreateLogger()) + .RunAsync(source, dest, plan, runTimeUtc, dryRun, reserved, Progress, pacer, ct) + .ConfigureAwait(false); + + // 5b. Final drain of the join projections to head (they're already caught up from pacing). + if (projectionCopy is not null) + { + Progress.EnterPhase(MigrationPhase.Draining); + await copier!.DrainAsync(copied, projectionCopy.DestProjections, dryRun, + projectionCopy.DrainTimeout, ct).ConfigureAwait(false); + } + + if (copy.UnparseableVerbatim > 0) + logger.LogWarning("{Count} source event(s) had unparseable JSON payloads and were copied " + + "VERBATIM (byte-for-byte, not dropped) — no data loss.", copy.UnparseableVerbatim); + + // 5c. INDEX-BASED merge recovery (optional, the clean alternative to the projection copy): now that the + // dest holds every aggregate stream, create one user-defined index per app merge on the dest, + // each read in commit order via $idx-user-… (no paced projection, no type-clustering). + IReadOnlyList createdIndexes = []; + if (indexCopy is not null) + { + Progress.EnterPhase(MigrationPhase.ProjectionCopy); + createdIndexes = await new UserDefinedIndexMergeBuilder(_lf) + .BuildAsync(indexCopy, dest, dryRun, ct).ConfigureAwait(false); + logger.LogInformation("Index copy: {Count} user-defined index(es) {Verb}: {Names}", + createdIndexes.Count, dryRun ? "planned (dry run — not created)" : "created on dest", + createdIndexes.Count == 0 ? "(none)" : string.Join(", ", createdIndexes.Select(i => i.IndexName))); + } + + if (dryRun) + { + Progress.Complete(DateTime.UtcNow); + return new MigrationRunResult + { + DryRun = true, + PendingMigrationIds = pending.Select(p => p.Id).ToList(), + Copy = copy, + NewlyApplied = [], + CopiedProjections = copied.Select(p => p.Name).ToList(), + CreatedIndexes = createdIndexes + }; + } + + // 6. Record history: existing (carried forward) + newly applied. + Progress.EnterPhase(MigrationPhase.RecordingHistory); + var newlyApplied = pending.Select(p => new MigrationApplied + { + Id = p.Id, + Name = p.Name, + Checksum = p.Checksum, + AppliedAtUtc = runTimeUtc, + SourceEvents = copy.SourceEvents, + Kept = copy.Kept, + Dropped = copy.MigrationStats[p.Id].Dropped, + Transformed = copy.MigrationStats[p.Id].Transformed + copy.MigrationStats[p.Id].Renamed, + Descriptors = plan.DescriptorsOf(p.Id) + }).ToList(); + await history.WriteAsync(dest, applied, newlyApplied, runTimeUtc, ct).ConfigureAwait(false); + Progress.MarkPendingCompleted(DateTime.UtcNow); + + // 7. Verify the destination against the copy bookkeeping. + Progress.EnterPhase(MigrationPhase.Verifying); + var verification = await new Verifier().VerifyAsync(dest, copy, reserved, ct).ConfigureAwait(false); + logger.LogInformation("{Report}", verification.Format()); + + Progress.Complete(DateTime.UtcNow); + return new MigrationRunResult + { + DryRun = false, + PendingMigrationIds = pending.Select(p => p.Id).ToList(), + Copy = copy, + Verification = verification, + NewlyApplied = newlyApplied, + CopiedProjections = copied.Select(p => p.Name).ToList(), + CreatedIndexes = createdIndexes + }; + } + catch (Exception ex) + { + Progress.Fail(ex.Message, DateTime.UtcNow); + throw; + } + } + + /// Count of the leading run of ASCII digits in an id (e.g. "0012_foo" → 4, "foo" → 0). + private static int LeadingDigitWidth(string id) + { + var i = 0; + while (i < id.Length && id[i] is >= '0' and <= '9') i++; + return i; + } +} diff --git a/src/MicroPlumberd.Migration/ProjectionCopier.cs b/src/MicroPlumberd.Migration/ProjectionCopier.cs new file mode 100644 index 0000000..b8ee03e --- /dev/null +++ b/src/MicroPlumberd.Migration/ProjectionCopier.cs @@ -0,0 +1,254 @@ +using System.Security.Cryptography; +using System.Text; +using System.Text.Json; +using System.Text.RegularExpressions; +using KurrentDB.Client; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MicroPlumberd.Migration; + +/// +/// A user (non-$) continuous projection discovered on the source. is the VERBATIM +/// query text; and are parsed from it so the paced copy can +/// wait on the exact link emission. +/// +public sealed record ProjectionInfo(string Name, string Query, string OutputStream, IReadOnlySet LinkTypes); + +/// Callback the copy engine invokes after each kept event is appended, to PACE the copy. +internal interface IEventPacer +{ + /// Blocks until the join projection(s) have processed the just-appended event (its link emitted). + Task AfterAppendedAsync(string destType, CancellationToken ct); +} + +/// +/// Reinstates the owner's design: pre-create the app's linkTo JOIN projections on the EMPTY destination +/// so THEY are the sole writers of every [OutputStream] merge stream, then PACE the event copy so the +/// projection is kept caught up THROUGHOUT — it never accumulates a backlog, so a fromStreams catch-up +/// can never type-cluster (that clustering is a property of catching up over a multi-$et BACKLOG). +/// +/// +/// Pre-create: discover source user projections (skip $-system), read each query VERBATIM over the +/// HTTP projections API (the gRPC client exposes no query text), replicate MicroPlumberd's create flow +/// (Create→Disable→Update(emit)→Enable), and stamp mp_query_hash = SHA256(query) on the projection's +/// stream so the app's TryCreateJoinProjection NO-OPs on boot. +/// Pace: the copy engine flushes ONE event at a time and calls ; +/// the pacer waits until the merge stream's link count advances for that event before the next is written, so +/// events are processed singly, in commit order. +/// +internal sealed class ProjectionCopier(ILogger? logger = null) +{ + private const string QueryHashMetadataKey = "mp_query_hash"; // MUST match KurrentDBProjectionManagementClientExtensions + + private readonly ILogger _logger = logger ?? NullLogger.Instance; + + /// Lists the source's user (non-$) continuous projections, reading each query VERBATIM over HTTP. + public async Task> DiscoverAsync( + KurrentDBProjectionManagementClient sourceProjections, + string sourceConnectionString, + CancellationToken ct = default) + { + var (baseUri, user, pass) = KurrentHttpEndpoint.Parse(sourceConnectionString); + using var http = KurrentHttpEndpoint.CreateClient(user, pass); + + var result = new List(); + await foreach (var p in sourceProjections.ListContinuousAsync().WithCancellation(ct).ConfigureAwait(false)) + { + if (string.IsNullOrEmpty(p.Name) || p.Name[0] == '$') continue; + var query = await ReadQueryVerbatimAsync(http, baseUri, p.Name, ct).ConfigureAwait(false); + var linkTypes = ParseLinkTypes(query); + var outputStream = ParseOutputStream(query) ?? p.Name; + result.Add(new ProjectionInfo(p.Name, query, outputStream, linkTypes)); + _logger.LogInformation("Discovered user projection '{Name}' → output '{Out}', link types [{Types}].", + p.Name, outputStream, string.Join(",", linkTypes)); + } + return result; + } + + /// Verifies the destination's standard $by_event_type projection is running (required to feed the joins). + public async Task EnsureStandardProjectionsRunningAsync( + KurrentDBProjectionManagementClient destProjections, bool dryRun, CancellationToken ct = default) + { + const string sys = "$by_event_type"; + ProjectionDetails? status; + try { status = await destProjections.GetStatusAsync(sys, cancellationToken: ct).ConfigureAwait(false); } + catch (Exception ex) + { + throw new InvalidOperationException( + $"Destination must run standard projections ('{sys}'). Start the dest EventStore with " + + "KURRENTDB_RUN_PROJECTIONS=All and KURRENTDB_START_STANDARD_PROJECTIONS=true.", ex); + } + + if (IsRunning(status)) { _logger.LogInformation("Standard projection '{Sys}' is running.", sys); return; } + if (dryRun) { _logger.LogWarning("DRY RUN: '{Sys}' is '{Status}', not running.", sys, status?.Status); return; } + + await destProjections.EnableAsync(sys, cancellationToken: ct).ConfigureAwait(false); + status = await destProjections.GetStatusAsync(sys, cancellationToken: ct).ConfigureAwait(false); + if (!IsRunning(status)) + throw new InvalidOperationException( + $"Destination standard projection '{sys}' is '{status?.Status}' and could not be started."); + } + + /// Pre-creates each discovered projection on the (empty) dest as a live sole writer + stamps mp_query_hash. + public async Task PreCreateAsync( + IReadOnlyList projections, + KurrentDBProjectionManagementClient destProjections, + KurrentDBClient destClient, + bool dryRun, + CancellationToken ct = default) + { + foreach (var p in projections) + { + if (dryRun) + { + _logger.LogInformation("DRY RUN: would create projection '{Name}' and stamp mp_query_hash.", p.Name); + continue; + } + await destProjections.CreateContinuousAsync(p.Name, p.Query, trackEmittedStreams: false, + cancellationToken: ct).ConfigureAwait(false); + await destProjections.DisableAsync(p.Name, cancellationToken: ct).ConfigureAwait(false); + await destProjections.UpdateAsync(p.Name, p.Query, emitEnabled: true, cancellationToken: ct) + .ConfigureAwait(false); + await destProjections.EnableAsync(p.Name, cancellationToken: ct).ConfigureAwait(false); + await StoreQueryHashAsync(destClient, p.Name, ComputeQueryHash(p.Query), ct).ConfigureAwait(false); + _logger.LogInformation("Pre-created live projection '{Name}' (emit ON) + mp_query_hash.", p.Name); + } + } + + /// Builds the pacer that keeps the projections caught up event-by-event during the copy. + public IEventPacer CreatePacer(IReadOnlyList projections, KurrentDBClient destClient, + TimeSpan perEventTimeout) => + new ProjectionPacer(projections, destClient, perEventTimeout, _logger); + + /// Final drain: poll each projection to head so its checkpoint sits at the log head before boot. + public async Task DrainAsync(IReadOnlyList projections, + KurrentDBProjectionManagementClient destProjections, bool dryRun, TimeSpan timeout, + CancellationToken ct = default) + { + if (dryRun) return; + foreach (var p in projections) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + ct.ThrowIfCancellationRequested(); + var s = await destProjections.GetStatusAsync(p.Name, cancellationToken: ct).ConfigureAwait(false); + if (IsDrained(s)) break; + await Task.Delay(100, ct).ConfigureAwait(false); + } + } + } + + // ---- the pacer ----------------------------------------------------------------------------------------- + + private sealed class ProjectionPacer : IEventPacer + { + private readonly KurrentDBClient _dest; + private readonly TimeSpan _timeout; + private readonly ILogger _logger; + // Per projection: name, its output (merge) stream, its linkable dest-type set, and how many links waited-for. + private readonly List<(string Name, string Output, IReadOnlySet Types, long[] Expected)> _projections; + + public ProjectionPacer(IReadOnlyList projections, KurrentDBClient dest, TimeSpan timeout, + ILogger logger) + { + _dest = dest; + _timeout = timeout; + _logger = logger; + _projections = projections.Select(p => (p.Name, p.OutputStream, p.LinkTypes, new long[1])).ToList(); + } + + public async Task AfterAppendedAsync(string destType, CancellationToken ct) + { + foreach (var (name, output, types, expected) in _projections) + { + if (!types.Contains(destType)) continue; // this event is not linked into this merge stream + expected[0]++; + await WaitForLinkCountAsync(name, output, destType, expected[0], ct).ConfigureAwait(false); + } + } + + // Blocks until the merge stream has at least links (its last revision + 1), + // i.e. the projection has emitted the link for the just-appended event → it is caught up, no backlog. + // BOUNDED: if the projection does not emit within the timeout it FAILS LOUD (aborting the run) rather + // than hanging — a stalled/faulted projection mid-migration must not deadlock the copy. + private async Task WaitForLinkCountAsync(string projection, string outputStream, string destType, + long expected, CancellationToken ct) + { + var deadline = DateTime.UtcNow + _timeout; + while (true) + { + if (await LinkCountAsync(outputStream, ct).ConfigureAwait(false) >= expected) return; + if (DateTime.UtcNow >= deadline) + throw new TimeoutException( + $"Paced copy STALLED: projection '{projection}' did not emit link #{expected} into merge " + + $"stream '{outputStream}' for the just-appended '{destType}' event within " + + $"{_timeout.TotalSeconds:0}s. The projection is not keeping up (stopped/faulted?) — " + + "aborting the migration rather than proceeding with a possibly-reordered merge stream."); + await Task.Delay(15, ct).ConfigureAwait(false); + } + } + + // O(1): the stream's last event number + 1 (0 if it doesn't exist yet). + private async Task LinkCountAsync(string stream, CancellationToken ct) + { + var read = _dest.ReadStreamAsync(Direction.Backwards, stream, StreamPosition.End, maxCount: 1, + resolveLinkTos: false, cancellationToken: ct); + if (await read.ReadState.ConfigureAwait(false) == ReadState.StreamNotFound) return 0; + await foreach (var re in read.ConfigureAwait(false)) + return (long)re.OriginalEventNumber.ToUInt64() + 1; + return 0; + } + } + + // ---- helpers ------------------------------------------------------------------------------------------- + + private static bool IsRunning(ProjectionDetails? s) => + s?.Status is not null && s.Status.Contains("Running", StringComparison.OrdinalIgnoreCase); + + private static bool IsDrained(ProjectionDetails? s) => + s is not null && IsRunning(s) && s.Progress >= 99.9999f && s.BufferedEvents == 0 + && s.WritePendingEventsBeforeCheckpoint == 0 && s.WritePendingEventsAfterCheckpoint == 0; + + // Parse the linkable event types from a fromStreams(['$et-A','$et-B']) projection query. + internal static IReadOnlySet ParseLinkTypes(string query) + { + var set = new HashSet(StringComparer.Ordinal); + foreach (Match m in Regex.Matches(query, @"'\$et-([^']+)'")) + set.Add(m.Groups[1].Value); + return set; + } + + // Parse the linkTo('X', …) output stream target from the query (null if absent → caller falls back to name). + internal static string? ParseOutputStream(string query) + { + var m = Regex.Match(query, @"linkTo\(\s*'([^']+)'"); + return m.Success ? m.Groups[1].Value : null; + } + + private static string ComputeQueryHash(string query) => + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(query))); + + private static async Task StoreQueryHashAsync(KurrentDBClient esClient, string stream, string hash, + CancellationToken ct) + { + var existing = await esClient.GetStreamMetadataAsync(stream, cancellationToken: ct).ConfigureAwait(false); + var customDoc = JsonDocument.Parse($"{{\"{QueryHashMetadataKey}\":\"{hash}\"}}"); + // Fully qualify: core (referenced since the UserDefinedIndex relocation) also declares a + // MicroPlumberd.StreamMetadata that would otherwise shadow this KurrentDB.Client type. + var newMeta = new KurrentDB.Client.StreamMetadata(existing.Metadata.MaxCount, existing.Metadata.MaxAge, + existing.Metadata.TruncateBefore, existing.Metadata.CacheControl, existing.Metadata.Acl, customDoc); + await esClient.SetStreamMetadataAsync(stream, StreamState.Any, newMeta, cancellationToken: ct) + .ConfigureAwait(false); + } + + private static async Task ReadQueryVerbatimAsync(HttpClient http, Uri baseUri, string name, + CancellationToken ct) + { + var url = new Uri(baseUri, $"projection/{Uri.EscapeDataString(name)}/query"); + using var resp = await http.GetAsync(url, ct).ConfigureAwait(false); + resp.EnsureSuccessStatusCode(); + return await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + } +} diff --git a/src/MicroPlumberd.Migration/RawEvent.cs b/src/MicroPlumberd.Migration/RawEvent.cs new file mode 100644 index 0000000..3f74018 --- /dev/null +++ b/src/MicroPlumberd.Migration/RawEvent.cs @@ -0,0 +1,52 @@ +using System.Text.Json.Nodes; + +namespace MicroPlumberd.Migration; + +/// +/// A raw, non-typed view of a single stored event as seen by migration rules. +/// +/// Migrations NEVER deserialize into domain event classes — they operate purely on the event-type +/// string plus the JSON payload and metadata. This is what lets a migration rewrite events whose CLR +/// type has since been deleted or renamed: the library has zero reference to any app's event assemblies. +/// +/// +/// +/// is null when the stored payload is not application/json (e.g. a +/// binary / protobuf event) — such payloads are copied verbatim and cannot be transformed. +/// is a for MicroPlumberd-written events (it always +/// stores JSON metadata); it is null only when the source metadata is absent or non-JSON. +/// +public sealed record RawEvent +{ + /// The source stream the event was read from (e.g. Offer-of-…). + public required string StreamId { get; init; } + + /// The per-stream event number (revision) at the source, 0-based. + public required ulong EventNumber { get; init; } + + /// The stored event-type string (e.g. OfferCreated). + public required string Type { get; init; } + + /// + /// The JSON payload, or null when there is none to give: a payload that is not JSON, AND one that + /// claims to be JSON but does not parse (copied verbatim and counted as + /// CopyResult.UnparseableVerbatim). A rule is handed both cases and must expect null — a + /// caller that assumes otherwise meets a in the middle of a rewrite. + /// + public required JsonNode? Data { get; init; } + + /// The JSON metadata, or null when the source has none. + public required JsonNode? Metadata { get; init; } + + /// + /// The stored event id. Preserved verbatim by the copy engine, so a rule may match an event by the id an + /// operator read out of the source store (the id is the only stable per-event handle across a rewrite). + /// + public required Guid EventId { get; init; } + + /// + /// The instant the event was written at the SOURCE (UTC), so a rule may match by date. This is the source + /// record's own timestamp — the destination stamps its own Created on append. + /// + public required DateTime Created { get; init; } +} diff --git a/src/MicroPlumberd.Migration/UserDefinedIndexCopyContext.cs b/src/MicroPlumberd.Migration/UserDefinedIndexCopyContext.cs new file mode 100644 index 0000000..a4d1e87 --- /dev/null +++ b/src/MicroPlumberd.Migration/UserDefinedIndexCopyContext.cs @@ -0,0 +1,130 @@ +using System.Security.Cryptography; +using System.Text; +using KurrentDB.Client; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MicroPlumberd.Migration; + +/// +/// Enables the CLEAN index-based merge recovery (the alternative to ). When +/// supplied to , then — AFTER the plain copy has written every aggregate +/// stream to the fresh destination — the runner discovers the app's [OutputStream] join projections on +/// the source and creates one KurrentDB 26.1 USER-DEFINED INDEX per merge on the DESTINATION, filtering that +/// merge's event types. Each index is read via in true commit order +/// ($idx-user-{name}), so a consumer gets the merged stream WITHOUT the type-clustering that forces +/// to pace its copy — no paced projection, no backlog, no mp_query_hash. +/// +/// +/// ⚠ CONSUMER REWIRING REQUIRED — this is NOT a drop-in replacement for +/// . The index path does NOT rebuild the physical [OutputStream] +/// merge streams: no $> links are written into FooModel_v1 and no mp_query_hash is +/// stamped. A consumer that still subscribes to the physical merge stream will see it EMPTY (or, worse, the app +/// will re-create its own fromStreams join projection on boot and type-cluster it again). To use this +/// path a consumer MUST be rewired to read the index stream named in +/// ($idx-user-…) via a filtered $all read. If you need the physical merge stream rebuilt in place +/// for existing consumers, use instead. +/// This context and are mutually exclusive — supplying both to +/// throws. Omitting both runs the simple copy (merge streams skipped). +/// The index path creates nothing on the destination BEFORE the copy, so the copy's dest-empty pre-flight is +/// unaffected. +/// +public sealed class UserDefinedIndexCopyContext +{ + /// Projection-management client for the SOURCE store (used to discover the app's join projections). + public required KurrentDBProjectionManagementClient SourceProjections { get; init; } + + /// SOURCE connection string — used to read each projection's query verbatim over HTTP. + public required string SourceConnectionString { get; init; } + + /// DEST connection string — the indexes are created and read on the destination (it holds the copy). + public required string DestConnectionString { get; init; } + + /// Per-index readiness (create + backfill) timeout. Default 60s. + public TimeSpan ReadyTimeout { get; init; } = TimeSpan.FromSeconds(60); +} + +/// +/// A user-defined index created on the destination to serve one [OutputStream] merge in commit order. +/// The physical merge stream is NOT rebuilt — consumers must be rewired to +/// read instead (see ). +/// +/// The app merge/output stream this index REPLACES (e.g. FooModel_v1) — left EMPTY on the dest. +/// The KurrentDB index name (normalised + hash-disambiguated). +/// The read stream — $idx-user-{IndexName} — a consumer must read via filtered $all INSTEAD of . +/// The event types the index selects. +public sealed record CreatedIndex(string OutputStream, string IndexName, string IndexStream, + IReadOnlyCollection EventTypes); + +/// +/// Builds the destination user-defined indexes for : discovers the +/// app's join projections on the source (reusing ), then creates + +/// waits-ready one index per merge on the destination. +/// +internal sealed class UserDefinedIndexMergeBuilder(ILoggerFactory loggerFactory) +{ + private readonly ILoggerFactory _lf = loggerFactory ?? NullLoggerFactory.Instance; + + public async Task> BuildAsync( + UserDefinedIndexCopyContext ctx, KurrentDBClient dest, bool dryRun, CancellationToken ct = default) + { + var logger = _lf.CreateLogger(); + + // Reuse the ProjectionCopier discovery: it lists the source's user projections and parses each query's + // OutputStream + $et link types — exactly the (name, event-types) an index needs. + var copier = new ProjectionCopier(_lf.CreateLogger()); + var projections = await copier.DiscoverAsync(ctx.SourceProjections, ctx.SourceConnectionString, ct) + .ConfigureAwait(false); + + var source = new UserDefinedIndexSource(dest, ctx.DestConnectionString, + _lf.CreateLogger()); + + var created = new List(); + foreach (var p in projections) + { + if (p.LinkTypes.Count == 0) + { + logger.LogInformation("Projection '{Name}' selects no $et types — no index created.", p.Name); + continue; + } + + var indexName = IndexNameFor(p.OutputStream); + var spec = new UserDefinedIndexSpec { Name = indexName, EventTypes = p.LinkTypes }; + await source.EnsureAsync(spec, dryRun, ct).ConfigureAwait(false); + if (!dryRun) + { + // AUTHORITATIVE readiness: the dest holds the whole copy and its $all is immediately consistent, + // so the exact number of matching events is known — wait for the index to reach exactly that + // (never a stability heuristic that could return a truncated merge). + var expected = await source.CountMatchingAsync(p.LinkTypes, ct).ConfigureAwait(false); + await source.WaitUntilReadyAsync(indexName, expected, ctx.ReadyTimeout, ct).ConfigureAwait(false); + } + + var record = new CreatedIndex(p.OutputStream, indexName, + UserDefinedIndexSource.IndexStream(indexName), p.LinkTypes.ToArray()); + created.Add(record); + logger.LogInformation( + "Merge '{Out}' → user-defined index '{Index}' (read '{Stream}') over types [{Types}].", + p.OutputStream, indexName, record.IndexStream, string.Join(",", p.LinkTypes)); + } + + if (created.Count > 0) + logger.LogWarning( + "CONSUMER REWIRING REQUIRED: {Count} merge(s) are now served by user-defined indexes — the " + + "physical [OutputStream] merge stream(s) [{Streams}] are NOT rebuilt on the dest. Consumers " + + "must read the index stream(s) [{IndexStreams}] via a filtered $all read instead. Use " + + "ProjectionCopyContext if you need the physical merge stream rebuilt in place.", + created.Count, string.Join(", ", created.Select(c => c.OutputStream)), + string.Join(", ", created.Select(c => c.IndexStream))); + return created; + } + + // A valid, collision-resistant index name for a merge/output stream: the normalised name plus a short SHA-256 + // suffix of the ORIGINAL stream id, so two different streams that normalise to the same string never clash. + internal static string IndexNameFor(string outputStream) + { + var normalized = UserDefinedIndexSource.NormalizeName(outputStream); + var hash = Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(outputStream))).ToLowerInvariant(); + return $"mpidx-{normalized}-{hash[..8]}"; + } +} diff --git a/src/MicroPlumberd.Migration/UserDefinedIndexSource.cs b/src/MicroPlumberd.Migration/UserDefinedIndexSource.cs new file mode 100644 index 0000000..d148fb5 --- /dev/null +++ b/src/MicroPlumberd.Migration/UserDefinedIndexSource.cs @@ -0,0 +1,314 @@ +using System.Net; +using System.Runtime.CompilerServices; +using System.Text; +using System.Text.Json; +using System.Text.Json.Nodes; +using Grpc.Core; +using KurrentDB.Client; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MicroPlumberd.Migration; + +/// +/// Selects which stored events a indexes. is the +/// KurrentDB index name (lower-case alphanumerics, _ and - only — see +/// ); are the stored event-type +/// (schema) names to include. An EMPTY indexes ALL records. +/// +public sealed record UserDefinedIndexSpec +{ + /// The index name. Combined with the fixed prefix it forms the read stream $idx-user-{Name}. + public required string Name { get; init; } + + /// Stored event-type names to include; empty ⇒ index every record. + public required IReadOnlySet EventTypes { get; init; } +} + +/// +/// The CLEAN merged read-source for a migration (the alternative to ). It creates +/// a KurrentDB 26.1 USER-DEFINED INDEX over a set of event types and streams the matching events back in true +/// $all COMMIT order via the filtered-$all read (ReadAllAsync + +/// over $idx-user-{name} + resolveLinkTos). +/// +/// +/// Why this replaces the paced projection copy. An app-style +/// fromStreams(['$et-A','$et-B']).linkTo(X) join projection TYPE-CLUSTERS on catch-up (it drains +/// $et-A then $et-B → 0,2,4,1,3), which is why has to PACE the copy +/// event-by-event so a backlog never forms. A user-defined index is read through the filtered $all API, +/// which yields events in native commit order (0,1,2,3,4) with NO pacing and NO backlog games — exactly what a +/// migration merge needs. +/// Creation is idempotent. POSTs /v2/indexes/{name}; an existing +/// index (HTTP 409 INDEX_ALREADY_EXISTS, or an identical 200) is a no-op. If the existing index has a +/// DIFFERENT filter than requested it is left as-is and a warning is logged (KurrentDB does not redefine an +/// index in place). +/// Readiness (the NotFound/backfill gotcha). The index reports INDEX_STATE_STARTED +/// IMMEDIATELY after creation while it is still BACKFILLING history, and its resource may briefly 404 right +/// after the POST. therefore treats a 404 as "still building" (not "absent") +/// and, once the resource is STARTED, waits until the filtered read COUNT stops growing across consecutive +/// polls — i.e. the index has caught up to the (frozen, offline) source head — before the read is trusted to be +/// complete. Bounded by a timeout: a stalled build FAILS LOUD rather than yielding a short merge. +/// Every seam logs (index name + the failing URI on error) using the injected , +/// defaulting to so the type is safe to new up anywhere (incl. WASM). +/// +public sealed class UserDefinedIndexSource +{ + /// The read stream for an index is $idx-user-{name} — read it via filtered $all. + public const string IndexStreamPrefixRoot = UserDefinedIndex.IndexStreamPrefixRoot; + + private readonly KurrentDBClient _client; + private readonly Uri _httpBase; + private readonly string _user; + private readonly string _pass; + private readonly ILogger _logger; + private readonly UserDefinedIndex _core; + + /// + /// + /// gRPC client used for the filtered $all read. + /// + /// The node's esdb:// connection string — the HTTP management base and credentials are derived from it. + /// + /// Optional; defaults to . + public UserDefinedIndexSource(KurrentDBClient client, string connectionString, ILogger? logger = null) + { + _client = client ?? throw new ArgumentNullException(nameof(client)); + (_httpBase, _user, _pass) = KurrentHttpEndpoint.Parse(connectionString); + _logger = logger ?? NullLogger.Instance; + _core = new UserDefinedIndex(_httpBase, _user, _pass, _logger); + } + + /// The read stream ($idx-user-{name}) that carries 's indexed links. + public static string IndexStream(string name) => UserDefinedIndex.IndexStream(name); + + /// + /// Creates the index if it does not already exist (idempotent) — delegates to the core + /// . Returns without waiting for the backfill — call + /// before reading. No-op when is true. + /// + public Task EnsureAsync(UserDefinedIndexSpec spec, bool dryRun = false, CancellationToken ct = default) + { + ArgumentNullException.ThrowIfNull(spec); + return _core.EnsureAsync(spec.Name, spec.EventTypes, dryRun, ct); + } + + /// + /// Waits until the index has backfilled EXACTLY events — the AUTHORITATIVE + /// readiness signal, NOT a stability heuristic. is the known number of + /// matching events in the (frozen) store being indexed; obtain it from + /// against the same client (its $all is immediately consistent, so the count is exact, not a guess). + /// + /// + /// KurrentDB exposes no index checkpoint/progress (GET returns only name/filter/fields/state, and reports + /// STARTED before the backfill completes), so the only authoritative signal is the count of indexed + /// events converging to the KNOWN total. This method polls the filtered read until it returns + /// events: a NotFound (index still building) or a short count keeps waiting; + /// reaching the exact count is ready; EXCEEDING it throws (the index must never contain more than the store's + /// matching events). Bounded by → reporting how far + /// the backfill got, so a stalled/short build FAILS LOUD instead of returning a truncated merge. + /// + public async Task WaitUntilReadyAsync(string name, long expectedCount, TimeSpan timeout, + CancellationToken ct = default) + { + if (expectedCount < 0) throw new ArgumentOutOfRangeException(nameof(expectedCount)); + name = NormalizeName(name); + var deadline = DateTime.UtcNow + timeout; + var url = new Uri(_httpBase, $"v2/indexes/{Uri.EscapeDataString(name)}"); + using var http = KurrentHttpEndpoint.CreateClient(_user, _pass); + + // Phase 1: the index RESOURCE is present and STARTED (404 right after POST = still building, keep polling). + while (true) + { + ct.ThrowIfCancellationRequested(); + var state = await TryGetStateAsync(http, url, name, ct).ConfigureAwait(false); + if (state is not null && state.Contains("STARTED", StringComparison.OrdinalIgnoreCase)) break; + if (DateTime.UtcNow >= deadline) + throw new TimeoutException( + $"User-defined index '{name}' did not reach STARTED within {timeout.TotalSeconds:0}s " + + $"(last state '{state ?? ""}', {url})."); + await Task.Delay(100, ct).ConfigureAwait(false); + } + + // Phase 2: the BACKFILL has reached the KNOWN total. STARTED is reported IMMEDIATELY (before history is + // indexed), and the filtered read THROWS NotFound while the index is still building — that is "still + // building", NOT "empty" (the documented gotcha). Ready = the indexed count equals expectedCount exactly. + long last = -1; + while (true) + { + ct.ThrowIfCancellationRequested(); + var count = await TryCountAsync(name, ct).ConfigureAwait(false); + if (count is { } c) + { + last = c; + if (c == expectedCount) + { + _logger.LogInformation("User-defined index '{Name}' ready — {Count}/{Expected} event(s) indexed.", + name, c, expectedCount); + return; + } + if (c > expectedCount) + throw new InvalidOperationException( + $"User-defined index '{name}' indexed {c} events but only {expectedCount} were expected — " + + $"the filter matched more than the store's events ({IndexStream(name)}). This indicates a " + + "wrong expected count or an index-definition mismatch, not a transient state."); + } + if (DateTime.UtcNow >= deadline) + throw new TimeoutException( + $"User-defined index '{name}' backfill did not reach {expectedCount} event(s) within " + + $"{timeout.TotalSeconds:0}s (indexed {(last < 0 ? "" : last.ToString())} so far, " + + $"{IndexStream(name)}). The index is not keeping up (stopped/faulted?) — failing rather than " + + "proceeding with a truncated merge."); + await Task.Delay(100, ct).ConfigureAwait(false); + } + } + + /// + /// The AUTHORITATIVE count of events this index will contain: the number of events in this client's + /// $all whose stored type is in (or every non-system event when the set + /// is empty — the "index all" case). Because the migration creates the index AFTER the copy completes and + /// $all is immediately consistent, this is an EXACT expected total, not an estimate. Skips + /// $-system streams and $-typed events (link/tombstone markers never match a domain schema name). + /// + public async Task CountMatchingAsync(IReadOnlySet eventTypes, CancellationToken ct = default) + { + long n = 0; + await foreach (var re in _client.ReadAllAsync(Direction.Forwards, Position.Start, resolveLinkTos: false, + cancellationToken: ct).ConfigureAwait(false)) + { + var er = re.Event; + if (er is null) continue; + if (er.EventStreamId.Length > 0 && er.EventStreamId[0] == '$') continue; // system stream + if (er.EventType.Length > 0 && er.EventType[0] == '$') continue; // link/tombstone/system event + if (eventTypes.Count == 0 || eventTypes.Contains(er.EventType)) n++; + } + return n; + } + + /// + /// Streams the indexed events in true $all COMMIT order as s (payload/metadata + /// as raw JSON; is null for non-JSON payloads). Reads the RESOLVED target + /// events (resolveLinkTos), so / are + /// the original event's stream and revision. Call first for a complete read. + /// + public async IAsyncEnumerable ReadAsync(string name, [EnumeratorCancellation] CancellationToken ct = default) + { + name = NormalizeName(name); + var filter = StreamFilter.Prefix(IndexStream(name)); + var read = _client.ReadAllAsync(Direction.Forwards, Position.Start, filter, maxCount: long.MaxValue, + resolveLinkTos: true, cancellationToken: ct); + + await foreach (var re in read.ConfigureAwait(false)) + { + var er = re.Event; // resolved original (null if the link target is gone — skip, never dereference) + if (er is null) continue; + + yield return new RawEvent + { + StreamId = er.EventStreamId, + EventNumber = er.EventNumber.ToUInt64(), + Type = er.EventType, + Data = TryParseJson(er.Data), + Metadata = TryParseJson(er.Metadata), + EventId = er.EventId.ToGuid(), + Created = er.Created + }; + } + } + + /// + /// Convenience: → compute the authoritative expected count + /// () → → . + /// + public async IAsyncEnumerable CreateWaitReadAsync(UserDefinedIndexSpec spec, TimeSpan readyTimeout, + [EnumeratorCancellation] CancellationToken ct = default) + { + await EnsureAsync(spec, dryRun: false, ct).ConfigureAwait(false); + var expected = await CountMatchingAsync(spec.EventTypes, ct).ConfigureAwait(false); + await WaitUntilReadyAsync(spec.Name, expected, readyTimeout, ct).ConfigureAwait(false); + await foreach (var e in ReadAsync(spec.Name, ct).ConfigureAwait(false)) + yield return e; + } + + // ---- helpers ------------------------------------------------------------------------------------------- + + /// + /// Normalises a name to a valid KurrentDB index name — delegates to the core + /// (single implementation shared by the live and offline paths). Kept as a forwarder so existing internal + /// callers (e.g. ) are unchanged. + /// + internal static string NormalizeName(string raw) => UserDefinedIndex.NormalizeName(raw); + + /// Builds the index filter — delegates to the core . + internal static string? BuildFilter(IReadOnlySet eventTypes) => UserDefinedIndex.BuildFilter(eventTypes); + + private async Task TryGetStateAsync(HttpClient http, Uri url, string name, CancellationToken ct) + { + HttpResponseMessage resp; + try { resp = await http.GetAsync(url, ct).ConfigureAwait(false); } + catch (Exception ex) + { + _logger.LogWarning(ex, "GET user-defined index '{Name}' at {Uri} failed transiently — retrying.", name, url); + return null; + } + + if (resp.StatusCode == HttpStatusCode.NotFound) return null; // still building / not yet visible + if (!resp.IsSuccessStatusCode) + { + var body = await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + _logger.LogWarning("GET user-defined index '{Name}' at {Uri}: HTTP {Code} {Body}.", + name, url, (int)resp.StatusCode, body); + return null; + } + + var json = await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + var node = JsonNode.Parse(json); + return node?["index"]?["state"]?.GetValue(); + } + + // O(n) count of the indexed events EXACTLY as ReadAsync would yield them: same filtered read, same + // resolveLinkTos:true, same skip of a null-resolving (deleted-target) link — so the readiness gate count can + // NEVER diverge from what the reader emits (a resolveLinkTos:false count could reach the expected total while + // ReadAsync silently skips a dead link and yields fewer). Returns null when the index read stream does not + // yet exist — KurrentDB throws NotFound while the index is still BUILDING, which must read as "not ready", + // NOT "empty". n is the merge size (~thousands), so a full scan per poll is fine. + private async Task TryCountAsync(string name, CancellationToken ct) + { + var filter = StreamFilter.Prefix(IndexStream(name)); + var read = _client.ReadAllAsync(Direction.Forwards, Position.Start, filter, maxCount: long.MaxValue, + resolveLinkTos: true, cancellationToken: ct); + var e = read.GetAsyncEnumerator(ct); + long n = 0; + try + { + while (true) + { + try + { + if (!await e.MoveNextAsync().ConfigureAwait(false)) break; + } + catch (RpcException ex) when (ex.StatusCode == StatusCode.NotFound) + { + return null; // index read stream not materialised yet → still building + } + if (e.Current.Event is null) continue; // dead link — ReadAsync skips it, so it must not be counted + n++; + } + } + finally { await e.DisposeAsync().ConfigureAwait(false); } + return n; + } + + private static JsonNode? TryParseJson(ReadOnlyMemory bytes) + { + if (bytes.IsEmpty) return null; + try + { + var reader = new Utf8JsonReader(bytes.Span); + return JsonNode.Parse(ref reader); + } + catch (JsonException) + { + return null; + } + } +} diff --git a/src/MicroPlumberd.Migration/Verifier.cs b/src/MicroPlumberd.Migration/Verifier.cs new file mode 100644 index 0000000..e4d4d53 --- /dev/null +++ b/src/MicroPlumberd.Migration/Verifier.cs @@ -0,0 +1,161 @@ +using System.Security.Cryptography; +using System.Text; +using KurrentDB.Client; + +namespace MicroPlumberd.Migration; + +/// Whether a destination stream matched its expected count AND write-fidelity checksum. +public enum VerificationStatus +{ + /// Destination count AND checksum equal the expected (copied) values. + Ok, + + /// Destination count or checksum differs from expected and is not an intended drop. + Mismatch +} + +/// Per destination-stream verification outcome. +/// The destination stream. +/// Events the copy engine expected to write to this stream. +/// Events actually found in the destination stream. +/// The last event number (revision) in the destination stream. +/// Whether the dest stream's recomputed write-fidelity checksum matched the copy's. +/// Whether the counts AND checksum matched. +public sealed record StreamVerification( + string DestStream, + long ExpectedCount, + long ActualCount, + ulong FinalVersion, + bool HashOk, + VerificationStatus Status); + +/// A source stream that produced no destination events (an intended full drop). +/// The fully-dropped source stream. +/// How many events were dropped from it. +public sealed record DroppedStreamInfo(string SourceStream, long SourceCount); + +/// +/// Cross-checks the destination store against what the copy engine believes it wrote — both per-stream event +/// COUNTS and a per-stream write-fidelity CHECKSUM (recomputed from the dest). It verifies the copy INTENT +/// reached the dest faithfully (no dest-side reorder/truncation/corruption); it does NOT re-derive events from +/// the source or re-check transform semantics (those are covered by tests). Mismatches not explained by an +/// intended drop are flagged. +/// +public sealed class VerificationReport +{ + /// Per destination-stream verification rows. + public required IReadOnlyList Streams { get; init; } + + /// Source streams that were fully dropped (intended). + public required IReadOnlyList DroppedStreams { get; init; } + + /// True when every destination stream matches its expected count. + public bool AllOk => Streams.All(s => s.Status == VerificationStatus.Ok); + + /// Renders a human-readable report (per-stream expected vs actual counts and final versions). + public string Format() + { + var sb = new StringBuilder(); + sb.AppendLine("Verification report (source vs destination):"); + sb.AppendLine($" Destination streams: {Streams.Count}"); + foreach (var s in Streams.OrderBy(s => s.DestStream, StringComparer.Ordinal)) + { + var flag = s.Status == VerificationStatus.Ok ? "OK " : "!! "; + var hash = s.HashOk ? "" : " [CHECKSUM MISMATCH]"; + sb.AppendLine( + $" {flag}{s.DestStream}: expected {s.ExpectedCount}, actual {s.ActualCount}, final v{s.FinalVersion}{hash}"); + } + if (DroppedStreams.Count > 0) + { + sb.AppendLine($" Fully-dropped source streams (intended): {DroppedStreams.Count}"); + foreach (var d in DroppedStreams.OrderBy(d => d.SourceStream, StringComparer.Ordinal)) + sb.AppendLine($" - {d.SourceStream} ({d.SourceCount} event(s) dropped)"); + } + sb.AppendLine(AllOk ? " RESULT: OK" : " RESULT: MISMATCH — see !! lines above."); + return sb.ToString(); + } +} + +internal sealed class Verifier +{ + public async Task VerifyAsync(KurrentDBClient dest, CopyResult copy, + IReadOnlySet reservedStreams, CancellationToken ct = default) + { + // Expected destination counts = kept events grouped by their target stream. + var expected = new Dictionary(StringComparer.Ordinal); + foreach (var info in copy.SourceStreams.Values) + { + if (info.Kept == 0) continue; + expected.TryGetValue(info.TargetStream, out var cur); + expected[info.TargetStream] = cur + info.Kept; + } + + // Actual destination counts + final revision, RE-READ independently from the destination log, plus a + // per-stream rolling write-fidelity hash recomputed over each event's (EventType || 0x00 || Data) in + // dest write order (M4). Reading $all forward yields each stream's events in event-number = write order, + // so the recomputed hash catches dest-side REORDER / TRUNCATION / CORRUPTION that a count check cannot. + // $>/`$`-typed events are skipped SYMMETRICALLY with the copy engine (it never wrote them). + var actual = new Dictionary(StringComparer.Ordinal); + var lastRev = new Dictionary(StringComparer.Ordinal); + var hashers = new Dictionary(StringComparer.Ordinal); + var read = dest.ReadAllAsync(Direction.Forwards, Position.Start, resolveLinkTos: false, + cancellationToken: ct); + await foreach (var re in read.ConfigureAwait(false)) + { + var er = re.Event; + if (er is null) continue; + if (er.EventStreamId.Length > 0 && er.EventStreamId[0] == '$') continue; + if (reservedStreams.Contains(er.EventStreamId)) continue; + if (er.EventType.Length > 0 && er.EventType[0] == '$') continue; + actual.TryGetValue(er.EventStreamId, out var c); + actual[er.EventStreamId] = c + 1; + lastRev[er.EventStreamId] = er.EventNumber.ToUInt64(); + FeedHash(hashers, er.EventStreamId, er.EventType, er.Data.Span); + } + + var actualHash = new Dictionary(StringComparer.Ordinal); + foreach (var (stream, h) in hashers) + { + actualHash[stream] = Convert.ToHexString(h.GetHashAndReset()); + h.Dispose(); + } + + var streams = new List(); + foreach (var name in expected.Keys.Union(actual.Keys)) + { + expected.TryGetValue(name, out var exp); + actual.TryGetValue(name, out var act); + lastRev.TryGetValue(name, out var fv); + // HashOk when the copy recorded an expected hash for this stream and the dest recomputation matches. + // Streams with no expected hash (none written, e.g. a fully-dropped stream) are vacuously hash-ok. + var hashOk = !copy.DestStreamHashes.TryGetValue(name, out var expHash) + || (actualHash.TryGetValue(name, out var actHash) && actHash == expHash); + var status = exp == act && hashOk ? VerificationStatus.Ok : VerificationStatus.Mismatch; + streams.Add(new StreamVerification(name, exp, act, fv, hashOk, status)); + } + + var dropped = copy.SourceStreams + .Where(kv => kv.Value.SourceCount > 0 && kv.Value.Kept == 0) + .Select(kv => new DroppedStreamInfo(kv.Key, kv.Value.Dropped)) + .ToList(); + + return new VerificationReport { Streams = streams, DroppedStreams = dropped }; + } + + private static readonly byte[] HashSeparator = [0x00]; + + private static void FeedHash(Dictionary hashers, string stream, string eventType, + ReadOnlySpan data) + { + if (!hashers.TryGetValue(stream, out var h)) + hashers[stream] = h = IncrementalHash.CreateHash(HashAlgorithmName.SHA256); + var max = Encoding.UTF8.GetMaxByteCount(eventType.Length); + byte[]? rented = max > 512 ? System.Buffers.ArrayPool.Shared.Rent(max) : null; + Span buf = rented ?? stackalloc byte[512]; + var n = Encoding.UTF8.GetBytes(eventType, buf); + h.AppendData(buf[..n]); + if (rented is not null) System.Buffers.ArrayPool.Shared.Return(rented); + h.AppendData(HashSeparator); + h.AppendData(data); + } +} diff --git a/src/MicroPlumberd.Migration/dev-log-userdefined-index.md b/src/MicroPlumberd.Migration/dev-log-userdefined-index.md new file mode 100644 index 0000000..49e5c35 --- /dev/null +++ b/src/MicroPlumberd.Migration/dev-log-userdefined-index.md @@ -0,0 +1,129 @@ +# Dev-log — User-Defined-Index merge read-path (feature/mp-userdefined-index) + +Audience: the reviewer/tester (task B2). This is the CLEAN alternative to the paced `ProjectionCopier`, +using KurrentDB 26.1 **user-defined indexes**. `ProjectionCopier` is kept intact — both strategies ship. + +## What problem this solves + +An app-style `fromStreams(['$et-A','$et-B']).linkTo(X)` join projection **type-clusters** on catch-up: it +drains `$et-A` then `$et-B`, so a merged read comes out `0,2,4,1,3` instead of commit order `0,1,2,3,4` +(proven in `ProjectionBehaviorSpikes`). `ProjectionCopier` works around it by PACING the copy event-by-event +so a backlog never forms. A **user-defined index** is read through the filtered-`$all` API, which yields +events in native commit order with **no pacing and no backlog** — exactly what a migration merge needs. + +## Files + +Added (`src/MicroPlumberd.Migration/`): +- `UserDefinedIndexSource.cs` — the reader (public). create → poll-until-ready → stream `RawEvent`s in commit order. +- `UserDefinedIndexCopyContext.cs` — runner-facing context + `CreatedIndex` record + `UserDefinedIndexMergeBuilder` + (discovers the app's join projections on the source, creates one index per merge on the dest). +- `KurrentHttpEndpoint.cs` — shared HTTP-endpoint/creds parser extracted from `ProjectionCopier` (DRY; + used by both the projection copier and the index source). + +Changed: +- `ProjectionCopier.cs` — now uses `KurrentHttpEndpoint` (its private `ParseHttpEndpoint`/`CreateHttpClient` removed). Behaviour unchanged. +- `MigrationRunner.cs` — `RunAsync` gained an `UserDefinedIndexCopyContext? indexCopy = null` parameter + (mutually exclusive with `projectionCopy` — supplying both throws); `MigrationRunResult.CreatedIndexes` added. + +Tests (`src/MicroPlumberd.Migration.Tests/`): +- `UserDefinedIndexIntegrationTests.cs` — 4 integration tests against real KurrentDB 26.1 (same + `EventStoreServer` Docker fixture as the rest; NEVER Testcontainers; no `ClearAllPools`). + +## Read-path design (verified empirically against KurrentDB 26.1.0.3443) + +1. **Create** — `POST /v2/indexes/{name}` with body `{"filter":"rec => rec.schema.name == \"A\" || rec.schema.name == \"B\"","start":true}`. + - The filter MUST be a single-arg JS arrow function (`rec => …`) or KurrentDB returns HTTP 400. + - Index names: **lower-case alphanumerics, `_`, `-` only** (else HTTP 400). `NormalizeName` lower-cases + + replaces every other char with `-`; the runner appends an 8-char SHA-256 suffix of the original stream + name for collision-safety. + - Idempotent: an identical re-POST is HTTP 200; an existing index is HTTP 409 `INDEX_ALREADY_EXISTS` → + treated as success (a differing filter is logged as drift, existing kept — KurrentDB won't redefine in place). +2. **Poll until ready (AUTHORITATIVE — expected-count, not a heuristic)** — `GET /v2/indexes/{name}` reports + `INDEX_STATE_STARTED` **immediately**, BEFORE the backfill completes, and exposes **no checkpoint/progress** + (I probed `GET`, `?stats`, `?details`, `/stats`, `/checkpoint` on 26.1.0.3443 — only name/filter/fields/state). + The gotcha: the filtered `$all` read of `$idx-user-{name}` **throws `RpcException NotFound` while the index is + still building** (NOT empty). So readiness is measured against the **known total**: + `WaitUntilReadyAsync(name, expectedCount, timeout)` — (phase 1) waits for the resource to be STARTED (404 GET + = still building); (phase 2) polls the filtered read, treating `NotFound`/short counts as "still building", + and returns only when the indexed count **equals `expectedCount` exactly** (exceeding it throws — the filter + matched more than the store holds). `expectedCount` comes from `CountMatchingAsync(eventTypes)`, an exact + count of matching events in the indexed client's `$all` (immediately consistent after the copy, so it is a + KNOWN total, not an estimate). Bounded by a timeout → `TimeoutException` reporting how far the backfill got, + so a stalled/short build FAILS LOUD instead of a truncated merge. This replaced the earlier stability + heuristic (B2 MUST-FIX #1). + - The gate count (`TryCountAsync`) uses the SAME read shape as `ReadAsync` — `resolveLinkTos:true` and skips a + null-resolving (dead-link) event — so the count the gate waits on can NEVER diverge from what the reader + yields (a `resolveLinkTos:false` count could hit the total while `ReadAsync` silently drops a dead link). + - Event-type names are JS-escaped in the filter (`BuildFilter`/`EscapeJsString`) so a name with a quote, + backslash or control char can't break out of the `rec.schema.name == "…"` literal. +3. **Read in commit order** — `ReadAllAsync(Direction.Forwards, Position.Start, StreamFilter.Prefix("$idx-user-{name}"), resolveLinkTos: true)` + → yields the RESOLVED original events as `RawEvent`s (StreamId/EventNumber = the source event's stream + + revision; Data/Metadata = raw JSON, Data null for non-JSON). This is true `$all` commit order — NOT type-clustered. + +`ILogger` at every seam (index name + the failing URI on error); defaults to `NullLogger` (WASM-safe). + +## Runner wiring + +`MigrationRunner.RunAsync(src, dst, migrations, dryRun, projectionCopy: null, indexCopy: ctx)`: +after the plain copy has written every aggregate stream to the fresh dest, `UserDefinedIndexMergeBuilder` +discovers the app's join projections on the source (reusing `ProjectionCopier.DiscoverAsync` → OutputStream + +`$et-` link types), and for each creates + waits-ready one user-defined index on the DEST filtering those +types. Result: `MigrationRunResult.CreatedIndexes` — each entry names the read stream (`$idx-user-…`) a +consumer subscribes to for that merge, in commit order. + +Boundary (honest) — **CONSUMER REWIRING REQUIRED** (B2 MUST-FIX #2, now surfaced loudly): the index path does +NOT physically rebuild the `[OutputStream]` merge streams (no `$>` links written, no `mp_query_hash`) — that is +`ProjectionCopier`'s job and it is unchanged. The index path's contract is "consumers read `$idx-user-{name}` +in commit order." A consumer left pointed at the physical merge stream will see it EMPTY. This is now called +out in three places: the `UserDefinedIndexCopyContext` XML doc (⚠ block), the `CreatedIndex` / +`MigrationRunResult.CreatedIndexes` docs, and a runtime `LogWarning` emitted by `UserDefinedIndexMergeBuilder` +listing each merge stream and its replacement `$idx-user-…` read stream. + +## Andon / environment notes for the tester + +- Test image `docker.kurrent.io/kurrent-latest/kurrentdb:latest` == **26.1.0.3443** — supports `/v2/indexes`. If a + future image is <26.1 and lacks `/v2/indexes`, that is an Andon: STOP (the create returns 404/unsupported). +- The suite spins one KurrentDB container per test (unique GUID names). Test (d) starts 3 (source + 2 dests). +- The NuGet feed (`nuget.modelingevolution.com`) was flaky; if `NU1301`, restore with + `-p:RestoreSources="https://api.nuget.org/v3/index.json;https://nuget.modelingevolution.com/v3/index.json"`. + +## Test coverage (what each asserts) + +- `Index_reads_interleaved_multitype_merge_in_commit_order_not_type_clustered` — (a)+(b) CORE: A0,B1,A2,B3,A4 + reads back `0,1,2,3,4`, explicitly NOT the type-clustered `0,2,4,1,3`. +- `Create_then_immediately_wait_ready_yields_complete_backfilled_read` — (a) ready-poll waits for the full + backfill (50 events) even called immediately after create (NotFound-while-building handled). +- `Ensure_is_idempotent_on_recreate` — (c) second identical `EnsureAsync` does not throw; read unchanged. +- `WaitUntilReady_is_authoritative_waits_for_exact_count_and_times_out_if_short` — B2 MUST-FIX #1 proof: + `CountMatchingAsync` returns the exact total; waiting for that total succeeds and reads all N in order; + waiting for N+1 (unreachable) `TimeoutException`s instead of falsely settling — proves the readiness is + authoritative, not a stability heuristic. +- `Large_merge_reads_back_full_count_in_commit_order_no_truncation` — B2 MUST-FIX #1 large-backfill proof: + 20,000 events interleaved across two types (backfill spans many poll intervals, ~30s), read back FULL count + in commit order `0..19999` — proves no truncated tail regardless of how the backfill paces. +- `Index_copy_matches_projection_copy_merged_order_and_faithful_aggregate_copy` — (d) a full migration with + `UserDefinedIndexCopyContext` verifies OK (faithful aggregate copy) and its dest index reproduces the SAME + merged commit order (`a-created,a-refined,b-created,b-refined`) that the `ProjectionCopyContext` path builds + into the physical `FooModel_v1` — the two strategies are cross-checked against one source. + +## Results + +Clean full run of `MicroPlumberd.Migration.Tests` against real KurrentDB 26.1.0.3443 (Docker), after the +environment restart, feed fallback applied: + +``` +Total tests: 43 Passed: 43 Failed: 0 Skipped: 0 (5.6 min) +``` + +- All 4 new `UserDefinedIndexIntegrationTests` PASS (commit-order 16s, ready-poll 15s, idempotent 15s, + index-vs-projection-copy 56s). +- The existing `IncidentReplayIntegrationTests` (which exercise the refactored `ProjectionCopier`) PASS, + including the 15/15 paced-copy determinism gate — `ProjectionCopier` is unregressed by the + `KurrentHttpEndpoint` extraction. +- Build: 0 errors (test project + all its ProjectReferences incl. `MicroPlumberd.Migration.Runner`, the one + consumer of the changed `RunAsync` signature — the new `indexCopy` param is optional and inserted before the + optional `ct`; no caller passed `ct` positionally). + +One bug was found and fixed DURING testing: the filtered `$all` read throws `RpcException NotFound` while the +index backfills (not empty) — readiness now handles it (see phase-2 above). It surfaced only under the full +parallel suite (slower backfill), not in isolation — a genuine heisenbug the single-test run hid. diff --git a/src/MicroPlumberd.Rewrite.Tests/CopyEngineTransformTests.cs b/src/MicroPlumberd.Rewrite.Tests/CopyEngineTransformTests.cs new file mode 100644 index 0000000..f1aff11 --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/CopyEngineTransformTests.cs @@ -0,0 +1,178 @@ +using System.Text; +using System.Text.Json.Nodes; +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Migration.Scripting; +using MicroPlumberd.Testing; +using Xunit; + +namespace MicroPlumberd.Rewrite.Tests; + +/// +/// UT-06/07 — the generic Transform operation inside the real copy engine, against real (in-memory) +/// KurrentDB containers. Exercised through the PUBLIC / +/// surface, exactly as the tool will use it. +/// +[Trait("Category", "Integration")] +public class CopyEngineTransformTests +{ + /// Provisions isolated source/destination stores and removes their containers afterwards. + private sealed class Stores : IAsyncDisposable + { + private readonly List _servers = []; + + public async Task NewStoreAsync(string tag) + { + // Named so a leftover container is obviously ours and can be removed by exact name — never by + // image or ancestor filter, which on this shared docker host would hit other sessions. + var srv = EventStoreServer.Create($"mp-rewrite-test-{tag}-{Guid.NewGuid():N}"); + _servers.Add(srv); + await srv.StartInDocker(inMemory: true); + return new KurrentDBClient(srv.GetEventStoreSettings()); + } + + public async ValueTask DisposeAsync() + { + foreach (var s in _servers) await s.DisposeAsync(); + } + } + + private sealed class InlineMigration(string id, Action build) : MicroPlumberd.Migration.Migration + { + public override string Id => id; + public override void Migrate(IMigrationBuilder b) => build(b); + } + + private static Task AppendAsync(KurrentDBClient c, string stream, StreamState expected, + string type, string dataJson, Uuid id) => + c.AppendToStreamAsync(stream, expected, + [new EventData(id, type, Encoding.UTF8.GetBytes(dataJson), Encoding.UTF8.GetBytes("{}"))]); + + private static async Task> ReadAsync(KurrentDBClient c, string stream) + { + var res = c.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: false); + if (await res.ReadState == ReadState.StreamNotFound) return []; + var list = new List(); + await foreach (var e in res) list.Add(e.Event); + return list; + } + + // ------------------------------------------------------------------ UT-06 + + [Fact] + public async Task UT06_Transform_retargets_the_stream_renumbers_gaplessly_preserves_event_ids_and_drops_on_null() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("ut06-src"); + var dst = await stores.NewStoreAsync("ut06-dst"); + + var ids = Enumerable.Range(0, 4).Select(_ => Uuid.NewUuid()).ToArray(); + // Deliberately NON-canonical JSON (a trailing .0 and spaces): a payload nothing touched must reach the + // destination byte-for-byte, not re-rendered. + await AppendAsync(src, "Src-1", StreamState.NoStream, "A", """{"n": 1.0}""", ids[0]); + await AppendAsync(src, "Src-1", 0ul, "Doomed", """{"n": 2}""", ids[1]); + await AppendAsync(src, "Src-1", 1ul, "A", """{"n": 3}""", ids[2]); + await AppendAsync(src, "Src-1", 2ul, "A", """{"n": 4}""", ids[3]); + var sourceBytes = (await ReadAsync(src, "Src-1")).ToDictionary(e => e.EventId, e => e.Data.ToArray()); + + var migration = new InlineMigration("0001_transform", b => b.Transform(e => + e.Type == "Doomed" ? null : e with { StreamId = "Dst-1" })); + + var result = await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: false); + + result.Copy.Dropped.Should().Be(1); + result.Copy.Kept.Should().Be(3); + + (await ReadAsync(dst, "Src-1")).Should().BeEmpty("every event was retargeted to Dst-1"); + + var copied = await ReadAsync(dst, "Dst-1"); + copied.Select(e => e.EventNumber.ToUInt64()).Should().Equal([0ul, 1ul, 2ul], + "the destination stream is renumbered gaplessly even though event #1 of the source was dropped"); + copied.Select(e => e.EventId).Should().Equal([ids[0], ids[2], ids[3]], + "the source event id is the only stable handle across a rewrite and must survive it"); + copied.Select(e => e.EventType).Should().Equal(["A", "A", "A"]); + copied[0].Data.ToArray().Should().Equal(sourceBytes[ids[0]], + "a payload no rule touched is copied byte-for-byte, not re-serialised"); + + // The history record must say WHAT the run did, not only that it ran — an operator reading a store + // months later has nothing else to go on. + var applied = result.NewlyApplied.Single(); + applied.Descriptors.Should().NotBeNull() + .And.ContainSingle(d => d.StartsWith("Transform("), + "the generic transform is the operation this migration consists of"); + } + + [Fact] + public async Task UT06_A_Transform_that_rewrites_the_payload_is_what_lands_in_the_destination() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("ut06b-src"); + var dst = await stores.NewStoreAsync("ut06b-dst"); + + var id = Uuid.NewUuid(); + await AppendAsync(src, "Order-1", StreamState.NoStream, "OrderCreated", + """{"Customer":"old","Keep":7}""", id); + + var migration = new InlineMigration("0001_payload", b => b.Transform(e => + { + var data = e.Data!.DeepClone(); + data["Customer"] = "X"; + return e with { Data = data, Type = "OrderCreatedV2" }; + })); + + await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: false); + + var copied = (await ReadAsync(dst, "Order-1")).Single(); + copied.EventType.Should().Be("OrderCreatedV2"); + copied.EventId.Should().Be(id); + var data = JsonNode.Parse(copied.Data.Span)!.AsObject(); + data["Customer"]!.GetValue().Should().Be("X"); + data["Keep"]!.GetValue().Should().Be(7, "a field the rule did not name must survive untouched"); + } + + // ------------------------------------------------------------------ UT-07 + + /// + /// A payload that claims to be JSON but is not cannot be handed to a script — there is no object to give + /// it. It bypasses the script, is copied byte-for-byte, and is COUNTED so an operator sees it in the + /// report. Remove the bypass and the script's o.Data.marker throws on undefined, so this + /// test goes red for exactly the reason its name claims. + /// + [Fact] + public async Task UT07_A_payload_that_is_not_JSON_bypasses_the_script_is_copied_verbatim_and_is_counted() + { + await using var stores = new Stores(); + var src = await stores.NewStoreAsync("ut07-src"); + var dst = await stores.NewStoreAsync("ut07-dst"); + + var junkId = Uuid.NewUuid(); + var goodId = Uuid.NewUuid(); + var junkBytes = "this is not json at all"u8.ToArray(); + // EventData defaults the content type to application/json — this is the "declared JSON, isn't" case + // the copy engine must never lose. + await src.AppendToStreamAsync("Blob-1", StreamState.NoStream, + [new EventData(junkId, "Blob", junkBytes, Encoding.UTF8.GetBytes("{}"))]); + await AppendAsync(src, "Order-1", StreamState.NoStream, "OrderCreated", """{"n":1}""", goodId); + + var migration = new ScriptMigration("function transform(o){ o.Data.marker = 1; return o; }", + "ut_20260907T000000"); + + var result = await new MigrationRunner().RunAsync(src, dst, [migration], dryRun: false); + + result.Copy.UnparseableVerbatim.Should().Be(1, "the operator must be told this event could not be transformed"); + result.Copy.Dropped.Should().Be(0, "an unreadable payload is never data loss"); + + var blob = (await ReadAsync(dst, "Blob-1")).Single(); + blob.Data.ToArray().Should().Equal(junkBytes); + blob.EventId.Should().Be(junkId); + + // The positive anchor: the very same script DID run, and did its work, on the JSON event next to it. + var order = (await ReadAsync(dst, "Order-1")).Single(); + JsonNode.Parse(order.Data.Span)!["marker"]!.GetValue().Should().Be(1); + + // The history carries the SCRIPT's hash, not a checksum over host-delegate IL — which every script + // shares, and which would therefore let a store rewritten by different rules look "already applied". + result.NewlyApplied.Single().Checksum.Should().Be(migration.ScriptChecksum); + } +} diff --git a/src/MicroPlumberd.Rewrite.Tests/DockerStoreTests.cs b/src/MicroPlumberd.Rewrite.Tests/DockerStoreTests.cs new file mode 100644 index 0000000..183062d --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/DockerStoreTests.cs @@ -0,0 +1,182 @@ +using Docker.DotNet.Models; +using FluentAssertions; +using MicroPlumberd.Rewrite; +using Xunit; + +namespace MicroPlumberd.Rewrite.Tests; + +/// UT-08 — resolving which mount holds the store, and refusing what cannot be swapped. +public class DockerStoreTests +{ + private static MountPoint Bind(string source, string destination) => + new() { Type = "bind", Source = source, Destination = destination }; + + private static MountPoint Volume(string name, string destination) => + new() { Type = "volume", Name = name, Source = $"/var/lib/docker/volumes/{name}/_data", Destination = destination }; + + [Fact] + public void UT08_The_env_configured_data_directory_wins_over_the_defaults() + { + // The trap this guards: a container told KURRENTDB_DB=/data that ALSO mounts something at the default + // path. Picking the default would rename the wrong directory away — and the backup would then be a + // backup of the wrong thing, which is the one mistake in this tool that cannot be undone by hand. + var location = DockerStore.ResolveDataLocation( + ["PATH=/usr/bin", "KURRENTDB_DB=/data"], + [Bind("/srv/other/data", "/var/lib/kurrentdb"), Bind("/srv/store/data", "/data")]); + + location.StoreDir.Should().Be("/srv/store/data"); + location.Destination.Should().Be("/data"); + location.IsBind.Should().BeTrue(); + } + + [Fact] + public void UT08_The_legacy_EVENTSTORE_DB_setting_is_honoured_too() + { + var location = DockerStore.ResolveDataLocation( + ["EVENTSTORE_DB=/var/lib/eventstore"], + [Bind("/srv/legacy/data", "/var/lib/eventstore")]); + + location.StoreDir.Should().Be("/srv/legacy/data"); + } + + [Fact] + public void UT08_With_no_env_setting_the_default_kurrentdb_path_is_used() + { + var location = DockerStore.ResolveDataLocation( + ["PATH=/usr/bin"], + [Bind("/var/docker/data/eventstore", "/var/lib/kurrentdb")]); + + location.StoreDir.Should().Be("/var/docker/data/eventstore"); + location.Parent.Should().Be("/var/docker/data"); + location.Name.Should().Be("eventstore", + "backups are named after the data directory, which is not always called 'data'"); + } + + [Fact] + public void UT08_A_named_volume_is_reported_as_such_and_carries_no_host_path() + { + var location = DockerStore.ResolveDataLocation([], [Volume("esdata", "/var/lib/kurrentdb")]); + + location.IsBind.Should().BeFalse(); + location.VolumeName.Should().Be("esdata"); + location.StoreDir.Should().BeNull("a named volume has no host path this tool may rename"); + } + + [Fact] + public void UT08_A_container_with_no_mount_at_its_data_directory_is_refused() + { + var act = () => DockerStore.ResolveDataLocation([], [Bind("/srv/logs", "/var/log/kurrentdb")]); + + act.Should().Throw() + .Where(e => e.Code == ExitCode.GuardRefusal) + .WithMessage("*no mount at its data directory*"); + } + + // ------------------------------------------------------------------ metrics parsing + + [Fact] + public void An_absent_connection_metric_reads_as_cannot_tell_not_as_zero() + { + // If a future KurrentDB renames the counter, "absent" must not silently become "no clients connected" + // — that would turn the guard off without anyone noticing, which is the failure it exists to prevent. + DockerStore.ReadMetric("some_other_metric{a=\"b\"} 3 1788\n", DockerStore.OpenGrpcCallsMetric) + .Should().Be(-1); + } + + [Fact] + public void The_open_grpc_call_count_is_read_from_a_labelled_prometheus_sample() + { + const string body = """ + # HELP kurrentdb_current_incoming_grpc_calls calls + kurrentdb_current_incoming_grpc_calls{otel_scope_name="KurrentDB.Core",otel_scope_version="1.0.0"} 2 1788792975770 + kurrentdb_kestrel_connections{otel_scope_name="KurrentDB.Core",otel_scope_version="1.0.0"} 3 1788792975770 + """; + + DockerStore.ReadMetric(body, DockerStore.OpenGrpcCallsMetric).Should().Be(2); + DockerStore.ReadMetric(body, DockerStore.KestrelConnectionsMetric).Should().Be(3); + } + + [Fact] + public void A_longer_metric_whose_name_merely_starts_the_same_is_not_mistaken_for_it() + { + // kurrentdb_incoming_grpc_calls_total sits right next to the gauge in a real /metrics body and starts + // with the same characters; matching it would report a lifetime TOTAL as a live count and refuse for ever. + const string body = "kurrentdb_current_incoming_grpc_calls_total{kind=\"total\"} 97 1788\n"; + + DockerStore.ReadMetric(body, DockerStore.OpenGrpcCallsMetric).Should().Be(-1); + } + + // ------------------------------------------------------------------ state paths outside the mount + + [Theory] + [InlineData("KURRENTDB_INDEX")] + [InlineData("KURRENTDB_DB")] + [InlineData("EVENTSTORE_INDEX")] + public void A_state_path_outside_the_swapped_mount_is_refused_and_the_setting_is_named(string key) + { + // The index is the dangerous one: the scratch store builds an index for the NEW log, but if the + // original container keeps its index somewhere this tool does not swap, it comes back on new data with + // a stale index — silent, and the same shape as the in-memory-database case. + var data = new DataLocation("/srv/store/data", "/var/lib/kurrentdb", IsBind: true, VolumeName: null); + + var (refusals, _) = DockerStore.CheckStatePathsInsideMount([$"{key}=/var/lib/kurrentdb-index"], data); + + refusals.Should().ContainSingle().Which.Should().Contain(key).And.Contain("/var/lib/kurrentdb"); + } + + [Fact] + public void A_state_path_inside_the_mount_is_accepted() + { + // The control: the gate must not refuse the ordinary layout, or it refuses every run. + var data = new DataLocation("/srv/store/data", "/var/lib/kurrentdb", IsBind: true, VolumeName: null); + + var (refusals, notes) = DockerStore.CheckStatePathsInsideMount( + ["KURRENTDB_DB=/var/lib/kurrentdb", "KURRENTDB_INDEX=/var/lib/kurrentdb/index"], data); + + refusals.Should().BeEmpty(); + notes.Should().BeEmpty(); + } + + [Fact] + public void A_log_path_outside_the_mount_is_reported_but_never_refused() + { + // Logs are not state. KurrentDB's own default log path is outside the data directory, so refusing on + // it would block a legitimate repair — with no override — for no safety gain. + var data = new DataLocation("/srv/store/data", "/var/lib/kurrentdb", IsBind: true, VolumeName: null); + + var (refusals, notes) = DockerStore.CheckStatePathsInsideMount(["KURRENTDB_LOG=/var/log/kurrentdb"], data); + + refusals.Should().BeEmpty("a log path cannot make a rewrite incorrect"); + notes.Should().ContainSingle().Which.Should().Contain("KURRENTDB_LOG"); + } + + [Theory] + [InlineData("/var/lib/kurrentdb", "/var/lib/kurrentdb", true)] + [InlineData("/var/lib/kurrentdb/index", "/var/lib/kurrentdb", true)] + [InlineData("/var/lib/kurrentdb-index", "/var/lib/kurrentdb", false)] + [InlineData("/var/lib/other", "/var/lib/kurrentdb", false)] + public void Containment_is_by_path_SEGMENT_not_by_string_prefix(string path, string root, bool inside) + { + // "/var/lib/kurrentdb-index".StartsWith("/var/lib/kurrentdb") is true and would silently accept a + // sibling directory that is not swapped at all. + DockerStore.IsInside(path, root).Should().Be(inside); + } + + // ------------------------------------------------------------------ scratch environment + + [Fact] + public void The_scratch_store_forces_projections_on_and_never_inherits_an_in_memory_database() + { + var env = DockerStore.BuildScratchEnv( + ["PATH=/usr/bin", "KURRENTDB_MEM_DB=true", "KURRENTDB_RUN_PROJECTIONS=None", "KURRENTDB_CLUSTER_SIZE=1"]); + + env.Should().Contain("KURRENTDB_RUN_PROJECTIONS=All") + .And.Contain("KURRENTDB_START_STANDARD_PROJECTIONS=true") + .And.Contain("KURRENTDB_INSECURE=true") + .And.Contain("KURRENTDB_MEM_DB=false", + "inheriting an in-memory database would copy the whole store into nothing and then swap it in"); + env.Should().Contain("KURRENTDB_CLUSTER_SIZE=1", "unrelated settings are carried over verbatim"); + env.Should().NotContain("KURRENTDB_RUN_PROJECTIONS=None") + .And.NotContain("KURRENTDB_MEM_DB=true", "a forced key REPLACES the old one rather than shadowing it"); + } +} diff --git a/src/MicroPlumberd.Rewrite.Tests/MicroPlumberd.Rewrite.Tests.csproj b/src/MicroPlumberd.Rewrite.Tests/MicroPlumberd.Rewrite.Tests.csproj new file mode 100644 index 0000000..7a3ad62 --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/MicroPlumberd.Rewrite.Tests.csproj @@ -0,0 +1,36 @@ + + + + net10.0 + enable + enable + false + true + + + + + + + runtime; build; native; contentfiles; analyzers; buildtransitive + all + + + + runtime; build; native; contentfiles; analyzers; buildtransitive + all + + + + + + + + + + + + + + + diff --git a/src/MicroPlumberd.Rewrite.Tests/RewriteCommandTests.cs b/src/MicroPlumberd.Rewrite.Tests/RewriteCommandTests.cs new file mode 100644 index 0000000..bf645fa --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/RewriteCommandTests.cs @@ -0,0 +1,90 @@ +using Docker.DotNet; +using FluentAssertions; +using MicroPlumberd.Rewrite; +using Microsoft.Extensions.Logging.Abstractions; +using Xunit; + +namespace MicroPlumberd.Rewrite.Tests; + +/// Command-level refusals that are decided before docker is consulted at all. +public class RewriteCommandTests +{ + private static RewriteOptions Options() => new() + { + // A container that does not exist. If the tool reached docker, the answer would be exit 3 + // "No such container" — so the assertions below double as proof that it did not. + Container = $"mp-rewrite-test-absent-{Guid.NewGuid():N}", + Yes = true, + Output = new StringWriter() + }; + + private static async Task RunAsync(RewriteOptions options) + { + using var docker = new DockerClientConfiguration().CreateClient(); + return await RewriteCommand.RunAsync(options, docker, NullLoggerFactory.Instance); + } + + [Theory] + [InlineData(RewriteMode.Rewrite)] + [InlineData(RewriteMode.Status)] + [InlineData(RewriteMode.Rollback)] + public async Task force_volume_copy_is_refused_with_exit_2_before_any_docker_call(RewriteMode mode) + { + // The ruling this pins: a flag the tool accepts and ignores is a trap. An operator on a named-volume + // host would pass it, be refused for the volume anyway, and have no way to tell the flag never helped. + var report = await RunAsync(Options() with { ForceVolumeCopy = true, Mode = mode }); + + // Exit 2: an ARGUMENT this version does not support. A named-volume STORE, without the flag, is a + // guard on the store's state and exits 1 — the test below pins that, so the pair cannot drift. + report.Code.Should().Be(ExitCode.ScriptError); + report.Headline.Should().Contain("--force-volume-copy", + "the refusal has to name the flag the operator typed") + .And.Contain("not implemented in this version") + .And.Contain("bind mount", "and say what to do instead"); + + report.Headline.Should().NotContain("No such container", + "reaching docker first would have produced this instead — the refusal must come before any call"); + report.Container.Should().BeNull("nothing was inspected"); + } + + [Fact] + public void A_store_on_a_named_volume_is_refused_with_exit_1() + { + var onAVolume = Container(new DataLocation(null, "/var/lib/kurrentdb", IsBind: false, VolumeName: "esdata")); + + var act = () => RewriteCommand.RequireSwappableData(onAVolume, Options()); + + act.Should().Throw() + .Where(e => e.Code == ExitCode.GuardRefusal, + "the STORE being unswappable is a guard on its state — exit 2 is for an unsupported argument") + .WithMessage("*esdata*", "the operator has to be told which volume") + .WithMessage("*bind mount*", "and what to do instead"); + } + + [Fact] + public void A_store_on_a_bind_mount_is_not_refused() + { + // The control: without it, "named volumes are refused" would also pass if everything were refused. + var onABind = Container(new DataLocation("/srv/store/data", "/var/lib/kurrentdb", IsBind: true, VolumeName: null)); + + var act = () => RewriteCommand.RequireSwappableData(onABind, Options()); + + act.Should().NotThrow(); + } + + private static StoreContainer Container(DataLocation data) => new() + { + Id = "abc123", Name = "some-store", Image = "kurrentdb:latest", Env = [], Data = data, Running = true + }; + + [Fact] + public async Task Without_the_flag_the_same_command_line_gets_as_far_as_docker() + { + // The control for the test above: without --force-volume-copy this exact invocation DOES reach docker. + // Without it, "the refusal came first" would be unfalsifiable — the tool might simply never call docker. + var report = await RunAsync(Options()); + + report.Code.Should().Be(ExitCode.DockerUnavailable); + report.Headline.Should().Contain("No such container"); + } +} diff --git a/src/MicroPlumberd.Rewrite.Tests/RewriteE2ETests.cs b/src/MicroPlumberd.Rewrite.Tests/RewriteE2ETests.cs new file mode 100644 index 0000000..6c5e244 --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/RewriteE2ETests.cs @@ -0,0 +1,836 @@ +using System.Text; +using System.Text.Json; +using System.Text.Json.Nodes; +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Rewrite; +using Microsoft.Extensions.Logging.Abstractions; +using Xunit; +using Xunit.Abstractions; + +namespace MicroPlumberd.Rewrite.Tests; + +/// +/// E2E-01..12 — the tool against real KurrentDB containers, per `iteration-1/test-scenarios.md`. +/// +/// +/// Every assertion reads the store through the ORIGINAL container after its restart. Reading the scratch +/// store would prove only that the copy engine can write — not that the operator ends up with the rewritten +/// history under the container they started with. +/// Each scenario disposes its seeding client before invoking the tool (E2E-09 excepted, where the open +/// subscription IS the scenario) — otherwise the connected-client guard would be catching the test itself. +/// +[Trait("Category", "Integration")] +[Collection("rewrite-e2e")] +public class RewriteE2ETests(ITestOutputHelper output) +{ + /// Uppercase hex SHA-256 of the empty string — the checksum a script-less run records. + private static readonly string EmptyScriptSha256 = + Convert.ToHexString(System.Security.Cryptography.SHA256.HashData([])); + + private static RewriteOptions Options(RewriteFixture f) => new() + { + Container = f.ContainerName, + Yes = true, + Output = new StringWriter() + }; + + private async Task RunAsync(RewriteFixture f, RewriteOptions options) + { + var report = await RewriteCommand.RunAsync(options, f.Docker, f.Loggers); + output.WriteLine(report.Format()); + return report; + } + + /// Runs a rewrite after making sure no client of the TEST's own is still holding the store. + private async Task RewriteAsync(RewriteFixture f, RewriteOptions options) + { + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + return await RunAsync(f, options); + } + + // ================================================================= E2E-01 + + [Fact] + public async Task E2E01_A_pure_copy_repairs_a_projection_faulted_by_a_dangling_link() + { + await using var f = await RewriteFixture.StartAsync(output); + await f.AssertDanglingLinkExistsAsync(); + + // The precondition the scenario is named for, established rather than assumed: an unguarded linkTo + // projection really is Faulted on this store, and the guarded one really is not. + var before = await f.WaitForFaultedAsync(RewriteFixture.FaultyProjection, TimeSpan.FromSeconds(60)); + before.Should().Contain("Faulted", + "the scenario is about repairing a FAULTED projection — if the fixture never faulted one, the " + + "'Running afterwards' assertion below would pass against a store that was never broken"); + (await f.ProjectionStatusAsync(RewriteFixture.MergeStream)).Should().Contain("Running"); + + var order1Before = await f.ReadAsync("Order-1"); + + var report = await RewriteAsync(f, Options(f)); + report.Code.Should().Be(ExitCode.Ok); + + // The container is back on P/data and the pre-rewrite store is kept beside it. + f.RootEntries().Should().Contain("data") + .And.ContainSingle(e => e.StartsWith("data.bak."), "the old store is never deleted"); + report.BackupDir.Should().NotBeNull(); + + // History survived byte-for-byte, read through the original container. + var order1After = await f.ReadAsync("Order-1"); + order1After.Select(e => e.EventType).Should().Equal(["OrderCreated", "LineAdded", "LineAdded"]); + order1After.Select(e => e.EventId).Should().Equal(order1Before.Select(e => e.EventId), + "a pure copy preserves every event id"); + order1After.Select(e => e.Data.ToArray()).Should() + .BeEquivalentTo(order1Before.Select(e => e.Data.ToArray()), o => o.WithStrictOrdering()); + + // The repair itself. + await using (var client = f.NewClient()) + { + var (total, dangling) = await RewriteFixture.CountLinksAsync(client, "$et-LineAdded"); + output.WriteLine($"after rewrite: $et-LineAdded has {total} link(s), {dangling} dangling"); + total.Should().BeGreaterThan(0, "zero links would satisfy 'none dangling' vacuously"); + dangling.Should().Be(0, "every link in the rewritten store must resolve"); + } + + var faultyAfter = await f.WaitForRunningAsync(RewriteFixture.FaultyProjection, TimeSpan.FromSeconds(60)); + faultyAfter.Should().Contain("Running").And.NotContain("Faulted", + "the projection that was Faulted on the old store runs on the rewritten one — that is the repair"); + (await f.ProjectionStatusAsync(RewriteFixture.MergeStream)).Should().Contain("Running"); + + // requirements.md § Safety and lead decision 4: EVERY run records history. A pure copy is the tool's + // headline invocation — the one an operator reaches for at 3 a.m. — and it must not be the one that + // leaves no trace that the store they are looking at is not the original. + var history = await f.ReadAsync("mp-migrations"); + history.Should().ContainSingle("a pure copy is still a run, and its history travels with the data"); + var applied = JsonSerializer.Deserialize(history[0].Data.Span)!; + applied.Id.Should().StartWith("rewrite_"); + applied.Checksum.Should().Be(EmptyScriptSha256, + "a pure copy applied no script, so the checksum is the hash of an empty one — which is honest, " + + "stays comparable with a scripted run, and is identical for every pure copy"); + applied.Descriptors.Should().NotBeNull().And.BeEmpty( + "'no rules were recorded' and 'no rules ran' must not look the same to whoever reads this later"); + applied.SourceEvents.Should().Be(report.SourceEvents).And.BeGreaterThan(0); + applied.Dropped.Should().Be(0); + + // uid 1001 in the container vs the operator's uid on the host: the swapped-in directory must stay + // writable by the container's user, and the only proof of that is a write. + await using (var client = f.NewClient()) + { + var write = async () => await client.AppendToStreamAsync("PostSwap-1", StreamState.NoStream, + [new EventData(Uuid.NewUuid(), "AfterSwap", Encoding.UTF8.GetBytes("""{"ok":true}"""))]); + await write.Should().NotThrowAsync( + "the container must still be able to write to the store it was handed, whatever uid it runs as"); + } + } + + // ================================================================= E2E-02 + + [Fact] + public async Task E2E02_dropStream_removes_the_named_streams_and_leaves_every_other_one_intact() + { + await using var f = await RewriteFixture.StartAsync(output); + var order1Before = await f.ReadAsync("Order-1"); + var order2Before = await f.ReadAsync("Order-2"); + (await f.ReadAsync("Junk-1")).Should().HaveCount(2, "the stream this scenario drops must be there first"); + + var report = await RewriteAsync(f, Options(f) with { Eval = "dropStream(/^Junk-/)" }); + + report.Code.Should().Be(ExitCode.Ok); + (await f.StreamExistsAsync("Junk-1")).Should().BeFalse("the dropped stream must not exist in the new store"); + report.DroppedStreams.Select(d => d.Stream).Should().Contain("Junk-1"); + + (await f.ReadAsync("Order-1")).Select(e => e.EventId).Should().Equal(order1Before.Select(e => e.EventId)); + (await f.ReadAsync("Order-2")).Select(e => e.EventId).Should().Equal(order2Before.Select(e => e.EventId)); + (await f.ReadAsync("Cmd-1")).Should().ContainSingle(); + } + + // ================================================================= E2E-03 + + [Fact] + public async Task E2E03_dropEvent_removes_one_event_and_the_survivors_are_renumbered_gaplessly() + { + await using var f = await RewriteFixture.StartAsync(output); + var before = await f.ReadAsync("Order-1"); + + // The stream is named in the rule so this variant and the transform variant below express exactly the + // same intent — `EventNumber === 1` alone would also hit Order-2's second event. + var report = await RewriteAsync(f, Options(f) with + { + Eval = """dropEvent(function(e){ return e.Stream === "Order-1" && e.EventType === "LineAdded" && e.EventNumber === 1; });""" + }); + + report.Code.Should().Be(ExitCode.Ok); + await AssertOrder1LostItsFirstLineAsync(f, before); + (await f.ReadAsync("Order-2")).Should().HaveCount(2, "a rule naming Order-1 must not touch Order-2"); + } + + [Fact] + public async Task E2E03_a_transform_returning_undefined_drops_the_same_event_as_dropEvent() + { + await using var f = await RewriteFixture.StartAsync(output); + var before = await f.ReadAsync("Order-1"); + + var report = await RewriteAsync(f, Options(f) with + { + Eval = """function transform(o){ return (o.Stream === "Order-1" && o.EventNumber === 1) ? undefined : o; }""" + }); + + report.Code.Should().Be(ExitCode.Ok); + await AssertOrder1LostItsFirstLineAsync(f, before); + } + + private static async Task AssertOrder1LostItsFirstLineAsync(RewriteFixture f, List before) + { + var after = await f.ReadAsync("Order-1"); + after.Select(e => e.EventNumber.ToUInt64()).Should().Equal([0ul, 1ul], + "the survivors are renumbered gaplessly — a gap would break every reader that expects a version"); + after.Select(e => e.EventType).Should().Equal(["OrderCreated", "LineAdded"]); + after.Select(e => e.EventId).Should().Equal([before[0].EventId, before[2].EventId], + "the surviving events keep their own ids; only the dropped one is gone"); + } + + // ================================================================= E2E-04 + + [Fact] + public async Task E2E04_update_by_type_and_by_id_change_only_what_the_script_named() + { + await using var f = await RewriteFixture.StartAsync(output); + + var report = await RewriteAsync(f, Options(f) with + { + Eval = $$""" + update("OrderCreated", function(e){ e.Data.Customer = "X"; return e; }); + updateById("{{RewriteFixture.CommandEventId.ToGuid()}}", function(e){ e.Metadata.Note = "n"; return e; }); + """ + }); + + report.Code.Should().Be(ExitCode.Ok); + + var order = (await f.ReadAsync("Order-1"))[0]; + var data = JsonNode.Parse(order.Data.Span)!.AsObject(); + data["Customer"]!.GetValue().Should().Be("X"); + data["Total"]!.GetValue().Should().Be(10, "a field the rule did not name must survive untouched"); + + var cmd = (await f.ReadAsync("Cmd-1")).Single(); + cmd.EventId.Should().Be(RewriteFixture.CommandEventId, "the event id is preserved across a rewrite"); + var meta = JsonNode.Parse(cmd.Metadata.Span)!.AsObject(); + meta["Note"]!.GetValue().Should().Be("n"); + meta["$correlationId"]!.GetValue().Should().Be("11111111-1111-1111-1111-111111111111"); + meta["$causationId"]!.GetValue().Should().Be("22222222-2222-2222-2222-222222222222"); + + // The by-type rule must not have leaked onto the command's payload. + JsonNode.Parse(cmd.Data.Span)!["Reason"]!.GetValue().Should().Be("manual"); + } + + // ================================================================= E2E-05 + + [Fact] + public async Task E2E05_a_script_file_containing_only_the_Replicator_transform_runs_unchanged() + { + await using var f = await RewriteFixture.StartAsync(output); + var scriptPath = Path.Combine(Path.GetTempPath(), $"mp-rewrite-{Guid.NewGuid():N}.js"); + // Nothing but `function transform(original)` — the contract Kurrent Replicator documents, with none + // of this library's helpers, so a script written for either tool runs on both. + await File.WriteAllTextAsync(scriptPath, """ + function transform(original) { + if (original.Stream === "Order-2") { + original.EventType = "V2." + original.EventType; + } + return original; + } + """); + try + { + var report = await RewriteAsync(f, Options(f) with { ScriptPath = scriptPath }); + + report.Code.Should().Be(ExitCode.Ok); + (await f.ReadAsync("Order-2")).Select(e => e.EventType) + .Should().Equal(["V2.OrderCreated", "V2.LineAdded"]); + (await f.ReadAsync("Order-1")).Select(e => e.EventType) + .Should().Equal(["OrderCreated", "LineAdded", "LineAdded"], "other streams are untouched"); + } + finally + { + File.Delete(scriptPath); + } + } + + // ================================================================= E2E-06 + + [Fact] + public async Task E2E06_a_dry_run_reports_what_it_would_do_and_leaves_everything_exactly_as_it_was() + { + await using var f = await RewriteFixture.StartAsync(output); + var startedAt = await f.StartedAtAsync(); + var junkBefore = await f.ReadAsync("Junk-1"); + + var report = await RewriteAsync(f, Options(f) with { Eval = "dropStream(\"Junk-1\")", DryRun = true }); + + report.Code.Should().Be(ExitCode.Ok); + report.DryRun.Should().BeTrue(); + report.DroppedStreams.Should().Contain(d => d.Stream == "Junk-1" && d.Events == 2, + "a dry run's whole value is naming the affected streams and their counts"); + + f.RootEntries().Should().Equal(["data"], + "a dry run leaves no new-store directory behind — not even the one it had to create to run"); + (await f.ScratchContainersAsync()).Should().BeEmpty("the scratch container is removed on a dry run too"); + (await f.StartedAtAsync()).Should().Be(startedAt, "a dry run never stops the container"); + + (await f.ReadAsync("Junk-1")).Select(e => e.EventId).Should().Equal(junkBefore.Select(e => e.EventId), + "nothing was written, so the store still holds the stream the rule would have dropped"); + } + + // ================================================================= E2E-07 + + [Fact] + public async Task E2E07_rollback_puts_the_pre_rewrite_store_back() + { + await using var f = await RewriteFixture.StartAsync(output); + var junkBefore = await f.ReadAsync("Junk-1"); + + (await RewriteAsync(f, Options(f) with { Eval = "dropStream(\"Junk-1\")" })).Code.Should().Be(ExitCode.Ok); + (await f.StreamExistsAsync("Junk-1")).Should().BeFalse("the rewrite must have taken effect first — " + + "rolling back a store that was never changed would prove nothing"); + + var report = await RewriteAsync(f, Options(f) with { Mode = RewriteMode.Rollback }); + + report.Code.Should().Be(ExitCode.Ok); + (await f.ReadAsync("Junk-1")).Select(e => e.EventId).Should().Equal(junkBefore.Select(e => e.EventId), + "the old history is back, event ids and all"); + f.RootEntries().Should().Contain(e => e.StartsWith("data.rolledback."), + "the rewritten store is kept, not deleted — a rollback must itself be reversible"); + (await f.ReadAsync("Order-1")).Should().HaveCount(3); + } + + // ================================================================= E2E-08 + + [Fact] + public async Task E2E08_a_running_sibling_of_the_same_compose_project_is_refused() + { + await using var f = await RewriteFixture.StartAsync(output); + var sibling = await f.StartSiblingAsync(); + var before = f.RootEntries(); + + var report = await RewriteAsync(f, Options(f)); + + report.Code.Should().Be(ExitCode.GuardRefusal); + report.Headline.Should().Contain(sibling, "the operator has to be told WHICH container to stop"); + f.RootEntries().Should().Equal(before, "a refusal creates nothing — not even the new-store directory"); + (await f.ScratchContainersAsync()).Should().BeEmpty("the guard runs before any container is started"); + (await f.ReadAsync("Order-1")).Should().HaveCount(3, "the store is untouched"); + } + + // ================================================================= E2E-09 + + [Fact] + public async Task E2E09_a_connected_client_is_refused_and_the_count_is_reported() + { + await using var f = await RewriteFixture.StartAsync(output); + var before = f.RootEntries(); + + // The one scenario that deliberately keeps a client open while the tool runs. + await using var client = f.NewClient(); + using var cts = new CancellationTokenSource(); + // Signalled once the subscription has actually been ESTABLISHED — the client connects lazily, so + // "the task was started" is not the same thing, and waiting on the wrong one is a race. + var established = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + var subscription = Task.Run(async () => + { + try + { + // From the START, so the fixture's existing events arrive at once: receiving one is direct + // evidence the subscription is live, and needs no assumption about which control messages this + // client surfaces through its default enumerator. + await foreach (var _ in client.SubscribeToAll(FromAll.Start, cancellationToken: cts.Token)) + established.TrySetResult(); + } + // The client surfaces its own cancellation as an RpcException(Cancelled), not as + // OperationCanceledException — catching only the latter turns tearing the subscription down into + // a test failure that says nothing about the tool. + catch (OperationCanceledException) { } + catch (Grpc.Core.RpcException ex) when (ex.StatusCode == Grpc.Core.StatusCode.Cancelled) { } + finally + { + // If it ended without ever being confirmed, unblock the wait below with the real reason + // instead of letting it time out saying something that is true but not the cause. + established.TrySetException(new InvalidOperationException( + "the $all subscription ended before it ever received an event")); + } + }); + + // 60s, not 30: establishing a subscription is slower when this shared host has just run sixty + // container-heavy scenarios, and the point of the wait is to reach the state the assertion needs. + await WaitUntilOpenCallsAsync(f, subscription, established.Task, atLeast: 1, TimeSpan.FromSeconds(60)); + + var report = await RunAsync(f, Options(f)); + + report.Code.Should().Be(ExitCode.GuardRefusal); + report.Headline.Should().Contain(DockerStore.OpenGrpcCallsMetric) + .And.Contain("gRPC call", "the refusal must say what it counted, not merely that it refused"); + f.RootEntries().Should().Equal(before); + (await f.ScratchContainersAsync()).Should().BeEmpty(); + + await cts.CancelAsync(); + await subscription; + } + + private async Task WaitUntilOpenCallsAsync(RewriteFixture f, Task subscription, Task established, + int atLeast, TimeSpan timeout) + { + // Wait for the subscription to be CONFIRMED by the server first. Polling the metric without this is a + // race — the subscribe call is only in flight once the client has actually connected, and under load + // that takes long enough for a naive poll to give up and blame the store. + var confirmed = await Task.WhenAny(established, subscription, Task.Delay(timeout)); + if (confirmed == established) await established; // rethrows the real reason if it failed + else if (confirmed == subscription) await subscription; // ended early — surface ITS exception + else throw new InvalidOperationException( + $"The $all subscription received no event within {timeout.TotalSeconds:0}s, so it was never live."); + + var deadline = DateTime.UtcNow + timeout; + var last = 0; + while (DateTime.UtcNow < deadline) + { + (last, _) = await DockerStore.ReadConnectionMetricsAsync(f.ConnectionString); + if (last >= atLeast) return; + if (subscription.IsCompleted) await subscription; + await Task.Delay(200); + } + throw new InvalidOperationException( + $"The subscription is live but the store reports {last} open gRPC call(s) (< {atLeast}); the " + + "scenario would otherwise 'pass' against a store with no client connected at all."); + } + + // ================================================================= E2E-10 + + [Fact] + public async Task E2E10_a_failure_after_the_swap_restores_the_original_store_and_restarts_the_container() + { + await using var f = await RewriteFixture.StartAsync(output); + var junkBefore = await f.ReadAsync("Junk-1"); + + var report = await RewriteAsync(f, Options(f) with + { + Eval = "dropStream(\"Junk-1\")", + FailAfterSwapForTest = true + }); + + report.Code.Should().Be(ExitCode.FilesystemFailure); + report.Headline.Should().Contain("restored"); + + // The whole point: the operator's data is what is live, not the half-finished rewrite. + (await f.ReadAsync("Junk-1")).Select(e => e.EventId).Should().Equal(junkBefore.Select(e => e.EventId), + "the ORIGINAL store is back in place, including the stream the rewrite would have dropped"); + (await f.ReadAsync("Order-1")).Should().HaveCount(3); + f.RootEntries().Should().Equal(["data"], + "the restore leaves no data.new.* and no data.bak.* — the rewrite is as if it had never run"); + + await using var client = f.NewClient(); + var write = async () => await client.AppendToStreamAsync("PostRestore-1", StreamState.NoStream, + [new EventData(Uuid.NewUuid(), "AfterRestore", Encoding.UTF8.GetBytes("""{"ok":true}"""))]); + await write.Should().NotThrowAsync("the container is running on the restored store and can write to it"); + } + + // ================================================================= exit 4 + + /// + /// M1 — the guards can only say "nobody is writing" at the instant they run, and between that instant and + /// the swap sit an unbounded confirmation prompt and the whole copy, with the old store up and writable. + /// An event appended in that window is not in the new store and the swap would destroy it, silently: the + /// engine never saw it, so its own bookkeeping balances perfectly without it. + /// + /// + /// The write happens while the tool is blocked on the confirmation prompt, which makes the race the + /// scenario is about deterministic — and needs no fault-injection hook, so what is exercised here is the + /// real trigger rather than a simulation of one. + /// + [Fact] + public async Task A_write_to_the_source_during_the_run_aborts_before_the_swap_with_exit_4() + { + await using var f = await RewriteFixture.StartAsync(output); + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + var before = f.RootEntries(); + var startedAt = await f.StartedAtAsync(); + + using var writesThenConfirms = AppendsThenConfirms(f, "LateWriter-1"); + var report = await RunAsync(f, Options(f) with + { + Yes = false, + ConfirmationInput = writesThenConfirms + }); + + writesThenConfirms.Acted.Should().BeTrue("the scenario is only meaningful if the write happened"); + report.Code.Should().Be(ExitCode.EngineFailure); + report.Headline.Should().Contain("written to DURING the run") + .And.Contain("old store is untouched"); + + // "Untouched" has to mean it: no swap, no backup, no scratch directory, container never stopped, and + // the late event still where the writer put it. + f.RootEntries().Should().Equal(before); + (await f.ScratchContainersAsync()).Should().BeEmpty(); + (await f.StartedAtAsync()).Should().Be(startedAt, "the container is never stopped before the gate passes"); + (await f.ReadAsync("LateWriter-1")).Should().ContainSingle("the write that caused the abort survived it"); + (await f.ReadAsync("Order-1")).Should().HaveCount(3); + } + + /// + /// Does something the moment the tool asks for confirmation, then answers it — which puts that something + /// inside the window between the guards and the swap, deterministically, with no fault injection. + /// + private sealed class ActOnPromptThenAnswer(Action act, string answer) : TextReader + { + public bool Acted { get; private set; } + + public override string ReadLine() + { + if (!Acted) { act(); Acted = true; } + return answer; + } + } + + private static ActOnPromptThenAnswer AppendsThenConfirms(RewriteFixture f, string stream) => + new(() => + { + using var client = f.NewClient(); + client.AppendToStreamAsync(stream, StreamState.Any, + [new EventData(Uuid.NewUuid(), "LateWrite", Encoding.UTF8.GetBytes("""{"late":true}"""))]) + .GetAwaiter().GetResult(); + }, "yes"); + + /// + /// The other half of M1: the guards are re-evaluated immediately before the swap. A sibling container + /// brought up during the confirmation prompt passed the first check and must still be caught — nothing + /// has been swapped yet, so refusing costs nothing. + /// + [Fact] + public async Task A_sibling_started_during_the_run_is_caught_by_the_guard_re_check_before_the_swap() + { + await using var f = await RewriteFixture.StartAsync(output); + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + var before = f.RootEntries(); + var startedAt = await f.StartedAtAsync(); + + string sibling = ""; + using var startsSiblingThenConfirms = new ActOnPromptThenAnswer( + () => sibling = f.StartSiblingAsync().GetAwaiter().GetResult(), "yes"); + + var report = await RunAsync(f, Options(f) with + { + Yes = false, + ConfirmationInput = startsSiblingThenConfirms + }); + + startsSiblingThenConfirms.Acted.Should().BeTrue(); + report.Code.Should().Be(ExitCode.GuardRefusal); + report.Headline.Should().Contain("last check before the swap").And.Contain(sibling); + + f.RootEntries().Should().Equal(before, "nothing was swapped and no backup was taken"); + (await f.ScratchContainersAsync()).Should().BeEmpty("the scratch store is removed on the way out"); + (await f.StartedAtAsync()).Should().Be(startedAt, "the container is never stopped"); + (await f.ReadAsync("Order-1")).Should().HaveCount(3); + } + + [Fact] + public async Task rollback_refuses_while_a_sibling_container_is_running() + { + // A rollback discards every write made since the backup, so it is MORE destructive than a rewrite — + // and it was the one path with no sibling and no connected-client check at all. + await using var f = await RewriteFixture.StartAsync(output); + (await RewriteAsync(f, Options(f) with { Eval = "dropStream(/^Junk-/)" })).Code.Should().Be(ExitCode.Ok); + (await f.StreamExistsAsync("Junk-1")).Should().BeFalse(); + + var sibling = await f.StartSiblingAsync(); + var report = await RunAsync(f, Options(f) with { Mode = RewriteMode.Rollback }); + + report.Code.Should().Be(ExitCode.GuardRefusal); + report.Headline.Should().Contain(sibling); + (await f.StreamExistsAsync("Junk-1")).Should().BeFalse( + "the rollback did not happen, so the rewritten store is still the live one"); + f.RootEntries().Should().NotContain(e => e.StartsWith("data.rolledback.")); + } + + // ================================================================= the confirmation prompt + + /// + /// The only human interlock in the tool, and the one branch that most needs to work: saying anything but + /// "yes" must leave everything exactly as it was. + /// + [Fact] + public async Task Declining_the_confirmation_changes_nothing() + { + await using var f = await RewriteFixture.StartAsync(output); + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + var before = f.RootEntries(); + var startedAt = await f.StartedAtAsync(); + var plan = new StringWriter(); + + var report = await RunAsync(f, Options(f) with + { + Yes = false, + Eval = "dropStream(/^Junk-/)", + ConfirmationInput = new StringReader("no\n"), + Output = plan + }); + + report.Code.Should().Be(ExitCode.GuardRefusal); + f.RootEntries().Should().Equal(before, "not even the new-store directory is created"); + (await f.ScratchContainersAsync()).Should().BeEmpty(); + (await f.StartedAtAsync()).Should().Be(startedAt, "the container is never stopped"); + (await f.ReadAsync("Junk-1")).Should().HaveCount(2, "the stream the rule would have dropped is still there"); + + // The prompt is also the only place the operator sees what the script was understood to MEAN. + var printed = plan.ToString(); + printed.Should().Contain("DropStreamPredicate", "the rules must be shown before they are applied"); + printed.Should().Contain(Convert.ToHexString( + System.Security.Cryptography.SHA256.HashData(Encoding.UTF8.GetBytes("dropStream(/^Junk-/)"))), + "and the checksum of the exact script text"); + } + + [Fact] + public async Task Confirming_with_yes_proceeds() + { + // The control: without this, "declining changes nothing" would also pass if the prompt rejected + // every answer, or if the tool never got past the prompt at all. + await using var f = await RewriteFixture.StartAsync(output); + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + + var report = await RunAsync(f, Options(f) with + { + Yes = false, + Eval = "dropStream(/^Junk-/)", + ConfirmationInput = new StringReader("yes\n") + }); + + report.Code.Should().Be(ExitCode.Ok); + (await f.StreamExistsAsync("Junk-1")).Should().BeFalse(); + } + + // ================================================================= --no-projection-copy + + [Fact] + public async Task no_projection_copy_suppresses_the_projection_copy_and_still_copies_the_events() + { + await using var f = await RewriteFixture.StartAsync(output); + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + + var report = await RunAsync(f, Options(f) with { NoProjectionCopy = true }); + + report.Code.Should().Be(ExitCode.Ok); + report.CopiedProjections.Should().BeEmpty(); + (await f.ProjectionStatusAsync(RewriteFixture.MergeStream)).Should().BeNull( + "with the copy suppressed the app's projection is not pre-created — the app recreates it on boot"); + (await f.ReadAsync("Order-1")).Should().HaveCount(3, "the events are copied either way"); + } + + // ================================================================= a second run + + /// + /// The positive anchor for every "no scratch container was left / started" assertion in this file. Those + /// are all negatives over a searchable space, and a query that can never see anything satisfies them for + /// free — proven: the reviewer replaced the filter with one that cannot match and they all stayed green. + /// + [Fact] + public async Task The_scratch_container_query_sees_the_scratch_container_while_it_is_up() + { + await using var f = await RewriteFixture.StartAsync(output); + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + + IReadOnlyList seenDuringRun = []; + // Look while the tool is blocked on the prompt: the scratch store is not up yet at that point, so the + // observation is taken from a background poll that runs across the whole rewrite instead. + using var polling = new CancellationTokenSource(); + var watcher = Task.Run(async () => + { + while (!polling.IsCancellationRequested) + { + var now = await f.ScratchContainersAsync(); + if (now.Count > 0) { seenDuringRun = now; return; } + await Task.Delay(200); + } + }); + + var report = await RunAsync(f, Options(f)); + await polling.CancelAsync(); + await watcher; + + report.Code.Should().Be(ExitCode.Ok); + seenDuringRun.Should().NotBeEmpty( + "if this query cannot see a scratch container that certainly existed, every 'no scratch container' " + + "assertion in this file is unfalsifiable"); + seenDuringRun.Should().AllSatisfy(n => n.Should().StartWith($"mp-rewrite-{f.ContainerName}-")); + (await f.ScratchContainersAsync()).Should().BeEmpty("and it is gone once the run finishes"); + } + + /// + /// RR-1 — the anchor for `--status`'s debris lines, against the PRODUCTION query rather than the + /// fixture's own. + /// + /// + /// The #4 anchor proves the FIXTURE's scratch-container helper can see something; it says nothing about + /// `DockerStore.FindScratchContainersAsync`, which is what `--status` actually calls. Same class of hole, + /// one level up: delete both status lines and nothing else in the suite goes red. + /// The debris is real, not fabricated — `--status` is run WHILE a rewrite is in flight, which is + /// exactly the state an operator sees when a run is interrupted: a `data.new.*` directory on disk and a + /// live `mp-rewrite-<ctr>-*` container. + /// + [Fact] + public async Task status_names_the_scratch_container_and_the_half_written_directory_while_a_rewrite_is_running() + { + await using var f = await RewriteFixture.StartAsync(output); + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + var production = new DockerStore(f.Docker, NullLogger.Instance); + + var rewrite = Task.Run(() => RewriteCommand.RunAsync(Options(f), f.Docker, f.Loggers)); + + // Wait for the state to exist, using the very method --status depends on. + IReadOnlyList liveScratch = []; + var deadline = DateTime.UtcNow + TimeSpan.FromSeconds(90); + while (DateTime.UtcNow < deadline && liveScratch.Count == 0 && !rewrite.IsCompleted) + { + liveScratch = await production.FindScratchContainersAsync(f.ContainerName); + if (liveScratch.Count == 0) await Task.Delay(200); + } + liveScratch.Should().NotBeEmpty( + "DockerStore.FindScratchContainersAsync must be able to SEE a scratch container that certainly " + + "exists — otherwise the --status line built on it is unfalsifiable"); + liveScratch.Should().AllSatisfy(n => n.Should().StartWith($"mp-rewrite-{f.ContainerName}-")); + + var during = (await RunAsync(f, Options(f) with { Mode = RewriteMode.Status })).Format(); + + during.Should().Contain(liveScratch[0], "--status must name the container an operator has to deal with"); + during.Should().Contain("data.new.", "and the half-written directory it cannot delete itself"); + during.Should().Contain("interrupted").And.Contain("scratch"); + + (await rewrite).Code.Should().Be(ExitCode.Ok); + + // The control: on a clean store both lines say so. Without it, "the lines appear" would also pass if + // they were hard-coded to appear always. + var after = (await RunAsync(f, Options(f) with { Mode = RewriteMode.Status })).Format(); + after.Should().Contain("interrupted : (none)").And.Contain("scratch : (none)"); + after.Should().NotContain("data.new."); + (await production.FindScratchContainersAsync(f.ContainerName)).Should().BeEmpty(); + } + + // ================================================================= E2E-11 + + [Fact] + public async Task E2E11_status_reports_the_store_its_backups_and_whether_a_rewrite_is_safe_now() + { + await using var f = await RewriteFixture.StartAsync(output); + await f.WaitUntilNoClientsAsync(TimeSpan.FromSeconds(30)); + var before = f.RootEntries(); + + var report = await RunAsync(f, Options(f) with { Mode = RewriteMode.Status }); + var text = report.Format(); + + report.Code.Should().Be(ExitCode.Ok); + report.Container!.Image.Should().Be(RewriteFixture.Image); + report.Container.Data.StoreDir.Should().Be(f.StoreDir); + text.Should().Contain("backups").And.Contain("(none)", "this store has never been rewritten"); + text.Should().Contain("safe : yes"); + + f.RootEntries().Should().Equal(before, "status changes nothing"); + (await f.ScratchContainersAsync()).Should().BeEmpty(); + + // And it says NO when a guard would refuse — the answer that actually matters to an operator. + var sibling = await f.StartSiblingAsync(); + var blocked = await RunAsync(f, Options(f) with { Mode = RewriteMode.Status }); + blocked.Format().Should().Contain("safe : no").And.Contain(sibling); + } + + // ================================================================= E2E-12 + + [Fact] + public async Task E2E12_the_new_store_carries_a_history_record_of_the_rewrite_that_produced_it() + { + await using var f = await RewriteFixture.StartAsync(output); + const string script = "dropStream(/^Junk-/)"; + + var report = await RewriteAsync(f, Options(f) with { Eval = script }); + report.Code.Should().Be(ExitCode.Ok); + + var history = await f.ReadAsync("mp-migrations"); + history.Should().ContainSingle("one run, one record"); + history[0].EventType.Should().Be(nameof(MigrationApplied)); + + var record = JsonSerializer.Deserialize(history[0].Data.Span)!; + record.Id.Should().Be(report.MigrationId); + record.Id.Should().StartWith("rewrite_", "each run is its own migration, stamped with when it ran"); + record.Checksum.Should().Be(Convert.ToHexString( + System.Security.Cryptography.SHA256.HashData(Encoding.UTF8.GetBytes(script))), + "the history carries the hash of the SCRIPT — the only thing that identifies these rules"); + record.Descriptors.Should().NotBeNull() + .And.Contain(d => d.StartsWith("DropStreamPredicate"), + "an operator reading this store months later must see WHAT the rewrite did"); + record.Dropped.Should().Be(2, "Junk-1 held two events"); + } + + [Fact] + public async Task A_second_run_appends_its_own_history_record() + { + // Lead decision 4's entire purpose: a second rewrite is a new migration, not one silently skipped as + // already applied. Two runs of the SAME script inside one second used to collide on the id. + await using var f = await RewriteFixture.StartAsync(output); + + (await RewriteAsync(f, Options(f))).Code.Should().Be(ExitCode.Ok); + var second = await RewriteAsync(f, Options(f)); + second.Code.Should().Be(ExitCode.Ok); + + var history = await f.ReadAsync("mp-migrations"); + history.Should().HaveCount(2, "the first run's record is carried forward and the second adds its own"); + var ids = history + .Select(e => JsonSerializer.Deserialize(e.Data.Span)!.Id) + .ToList(); + ids.Should().OnlyHaveUniqueItems("two runs must not share an id — the second would be skipped"); + ids.Should().AllSatisfy(id => id.Should().StartWith("rewrite_")); + } + + // ================================================================= exit codes + + [Fact] + public async Task A_script_that_does_not_parse_exits_2_before_any_container_is_touched() + { + await using var f = await RewriteFixture.StartAsync(output); + var before = f.RootEntries(); + var startedAt = await f.StartedAtAsync(); + + var report = await RunAsync(f, Options(f) with { Eval = "function transform(o) { return o;" }); + + report.Code.Should().Be(ExitCode.ScriptError); + report.Headline.Should().Contain("line").And.Contain("column"); + f.RootEntries().Should().Equal(before); + (await f.ScratchContainersAsync()).Should().BeEmpty("the script is parsed before docker is called at all"); + (await f.StartedAtAsync()).Should().Be(startedAt, "the container was never even stopped"); + } + + [Fact] + public async Task Naming_both_a_script_file_and_an_inline_script_exits_2() + { + // One invalid command line, one exit code. The parser used to reject this too, and returned 1 — + // which meant the code an operator's script branched on depended on which check ran first. + await using var f = await RewriteFixture.StartAsync(output); + + var report = await RunAsync(f, Options(f) with { Eval = "dropStream(\"a\")", ScriptPath = "/tmp/x.js" }); + + report.Code.Should().Be(ExitCode.ScriptError); + report.Headline.Should().Contain("mutually exclusive"); + } + + [Fact] + public async Task An_unknown_container_exits_3() + { + await using var f = await RewriteFixture.StartAsync(output); + + var report = await RunAsync(f, Options(f) with { Container = $"mp-rewrite-test-absent-{Guid.NewGuid():N}" }); + + report.Code.Should().Be(ExitCode.DockerUnavailable); + report.Headline.Should().Contain("No such container"); + } +} + +/// +/// Every end-to-end scenario runs alone. They start real containers on a shared docker host and the sibling +/// guard is about containers being up — two of them at once would refuse each other. +/// +[CollectionDefinition("rewrite-e2e", DisableParallelization = true)] +public sealed class RewriteE2ECollection; diff --git a/src/MicroPlumberd.Rewrite.Tests/RewriteFixture.cs b/src/MicroPlumberd.Rewrite.Tests/RewriteFixture.cs new file mode 100644 index 0000000..73b9487 --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/RewriteFixture.cs @@ -0,0 +1,488 @@ +using System.Text; +using System.Text.Json; +using Docker.DotNet; +using Docker.DotNet.Models; +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Rewrite; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; +using Xunit.Abstractions; + +namespace MicroPlumberd.Rewrite.Tests; + +/// +/// The end-to-end fixture: a REAL KurrentDB container whose data directory is a bind-mounted temp directory, +/// seeded exactly as `iteration-1/test-scenarios.md` describes, with the dangling link the pure-copy scenario +/// depends on. +/// +/// +/// The fixture proves its own precondition. A scenario about repairing a dangling link is +/// worthless if the fixture never produced one, and "the rewritten store has no dangling links" passes +/// trivially against a source that had none. is called before every +/// such scenario and fails the test if the setup did not take. +/// The container is labelled com.docker.compose.project=mp-rewrite-test so the sibling guard has +/// something real to find, and named mp-rewrite-test-* so a leftover is unmistakably ours and can be +/// removed by exact name. +/// +public sealed class RewriteFixture : IAsyncDisposable +{ + /// + /// The compose project this fixture's containers are labelled with. Per fixture rather than shared, so a + /// container left behind by a crashed run cannot make the next scenario refuse for someone else's reason. + /// + public string ComposeProject { get; } + + /// The event id `Cmd-1`'s only event carries — E2E-04 matches on it. + public static readonly Uuid CommandEventId = Uuid.FromGuid(Guid.Parse("c1ceefa2-363b-44ce-93a8-c54a0517550f")); + + /// The merge stream AND the projection name — MicroPlumberd names a join projection after its output. + public const string MergeStream = ">Test"; + + /// The stream whose events expire, leaving `$et-*` holding links that no longer resolve. + public const string ExpiringStream = "Expiring-1"; + + /// + /// A join projection written WITHOUT MicroPlumberd's null guard — the shape that faults on a dangling + /// link, which is the production incident this tool exists for (a command-router projection Faulted on the + /// AMZ neuron, four hours over VPN). + /// + public const string FaultyProjection = ">Faulty"; + + private readonly DockerClient _docker; + private readonly ILoggerFactory _loggerFactory; + private readonly ITestOutputHelper? _output; + private readonly List _extraContainers = []; + + private RewriteFixture(DockerClient docker, ILoggerFactory lf, ITestOutputHelper? output, string rootDir, + string storeDir, string containerId, string containerName, int port, string composeProject) + { + ComposeProject = composeProject; + _docker = docker; + _loggerFactory = lf; + _output = output; + RootDir = rootDir; + StoreDir = storeDir; + ContainerId = containerId; + ContainerName = containerName; + Port = port; + } + + /// The bind-mount parent — `P` in the scenarios. Backups and the new store appear here. + public string RootDir { get; } + + /// `P/data` — the live store directory. + public string StoreDir { get; } + + /// Docker id of the store container under test. + public string ContainerId { get; } + + /// Its name. + public string ContainerName { get; } + + /// The host port 2113 is published on (loopback only). + public int Port { get; } + + /// Connection string for the container under test. + public string ConnectionString => $"esdb://admin:changeit@127.0.0.1:{Port}?tls=false&tlsVerifyCert=false"; + + /// The image under test — override with MP_REWRITE_TEST_IMAGE. + public static string Image => + Environment.GetEnvironmentVariable("MP_REWRITE_TEST_IMAGE") + ?? "docker.kurrent.io/kurrent-latest/kurrentdb:latest"; + + /// Starts a container, seeds it, and proves the dangling-link precondition holds. + public static async Task StartAsync(ITestOutputHelper? output = null) + { + var docker = new DockerClientConfiguration().CreateClient(); + var lf = output is null + ? (ILoggerFactory)NullLoggerFactory.Instance + : LoggerFactory.Create(b => b.AddProvider(new TestOutputLoggerProvider(output)).SetMinimumLevel(LogLevel.Information)); + + var tag = Guid.NewGuid().ToString("N")[..10]; + var root = Path.Combine(Path.GetTempPath(), $"mp-rewrite-test-{tag}"); + var store = Path.Combine(root, "data"); + Directory.CreateDirectory(store); + // The image runs as uid 1001 and the test process does not; the store directory must be writable by + // the container's user or KurrentDB never starts. + DirectorySwap.MakeContainerWritable(root); + DirectorySwap.MakeContainerWritable(store); + + var name = $"mp-rewrite-test-{tag}"; + var composeProject = $"mp-rewrite-test-{tag}"; + var port = FreePort(); + + // A clean CI runner has no images at all. Pulling here rather than in the workflow keeps the suite + // self-sufficient wherever it runs, and puts the pull's time and failures in one obvious place instead + // of inside a container-start timeout. + await new DockerStore(docker, lf.CreateLogger("fixture")).EnsureImageAsync(Image); + var created = await docker.Containers.CreateContainerAsync(new CreateContainerParameters + { + Image = Image, + Name = name, + Labels = new Dictionary { [DockerStore.ComposeProjectLabel] = composeProject }, + Env = + [ + "KURRENTDB_RUN_PROJECTIONS=All", + "KURRENTDB_START_STANDARD_PROJECTIONS=true", + "KURRENTDB_INSECURE=true", + "KURRENTDB_ENABLE_ATOM_PUB_OVER_HTTP=true", + "KURRENTDB_MEM_DB=false" + ], + ExposedPorts = new Dictionary { ["2113"] = default }, + HostConfig = new HostConfig + { + // Default bridge only, published to loopback: this host's docker pool overlaps the company VPN, + // so a test must never create a network. + Binds = [$"{store}:/var/lib/kurrentdb"], + PortBindings = new Dictionary> + { + ["2113"] = [new PortBinding { HostIP = "127.0.0.1", HostPort = port.ToString() }] + } + } + }); + // Everything past this point can throw — the container may never become live, seeding may fail, the + // dangling-link precondition may not take. Without this, the container AND its root-owned data + // directory leak permanently: a uid-1000 operator cannot delete the directory (dev-log D11), and this + // host's root filesystem sits at 92 %. A failed SETUP has to clean up as reliably as a failed + // assertion already does. + var fixture = new RewriteFixture(docker, lf, output, root, store, created.ID, name, port, composeProject); + try + { + await docker.Containers.StartContainerAsync(created.ID, new ContainerStartParameters()); + await StoreHealth.WaitLiveAsync(fixture.ConnectionString, TimeSpan.FromSeconds(120), + lf.CreateLogger("fixture")); + await fixture.SeedAsync(); + return fixture; + } + catch + { + await fixture.DisposeAsync(); + throw; + } + } + + // ---------------------------------------------------------------- seeding + + private async Task SeedAsync() + { + var settings = KurrentDBClientSettings.Create(ConnectionString); + await using var client = new KurrentDBClient(settings); + await using var projections = new KurrentDBProjectionManagementClient(settings); + + await Append(client, "Order-1", StreamState.NoStream, ("OrderCreated", """{"Customer":"acme","Total":10}""")); + await Append(client, "Order-1", 0ul, ("LineAdded", """{"Sku":"a","Qty":1}""")); + await Append(client, "Order-1", 1ul, ("LineAdded", """{"Sku":"b","Qty":2}""")); + + await Append(client, "Order-2", StreamState.NoStream, ("OrderCreated", """{"Customer":"beta","Total":20}""")); + await Append(client, "Order-2", 0ul, ("LineAdded", """{"Sku":"c","Qty":3}""")); + + await Append(client, "Junk-1", StreamState.NoStream, ("Noise", """{"n":1}""")); + await Append(client, "Junk-1", 0ul, ("Noise", """{"n":2}""")); + + await client.AppendToStreamAsync("Cmd-1", StreamState.NoStream, + [ + new EventData(CommandEventId, "StopPipelineCommand", Encoding.UTF8.GetBytes("""{"Reason":"manual"}"""), + Encoding.UTF8.GetBytes( + """{"$correlationId":"11111111-1111-1111-1111-111111111111","$causationId":"22222222-2222-2222-2222-222222222222"}""")) + ]); + + // The dangling link: LineAdded events in a stream that then expires. $by_event_type has already linked + // them into $et-LineAdded, and those links outlive their targets. + await Append(client, ExpiringStream, StreamState.NoStream, ("LineAdded", """{"Sku":"ghost","Qty":9}""")); + await Append(client, ExpiringStream, 0ul, ("LineAdded", """{"Sku":"ghost2","Qty":8}""")); + await Append(client, ExpiringStream, 1ul, ("OrderCreated", """{"Customer":"ghost","Total":0}""")); + + await WaitForLinkCountAsync(client, "$et-LineAdded", 4, TimeSpan.FromSeconds(60)); + await WaitForLinkCountAsync(client, "$et-OrderCreated", 3, TimeSpan.FromSeconds(60)); + + // Expire them AFTER they are indexed, so the links exist and the targets stop resolving. + await client.SetStreamMetadataAsync(ExpiringStream, StreamState.Any, + new KurrentDB.Client.StreamMetadata(maxAge: TimeSpan.FromSeconds(1))); + + // The app's join projection: named after its output stream, exactly as MicroPlumberd creates it. + var query = $"fromStreams(['$et-OrderCreated']).when({{ $any: function(s,e){{ if(e && e.streamId !== null " + + $"&& e.sequenceNumber >= 0) linkTo('{MergeStream}', e) }} }});"; + await projections.CreateContinuousAsync(MergeStream, query, trackEmittedStreams: true); + await projections.DisableAsync(MergeStream); + await projections.UpdateAsync(MergeStream, query, emitEnabled: true); + await projections.EnableAsync(MergeStream); + await WaitForLinkCountAsync(client, MergeStream, 2, TimeSpan.FromSeconds(60)); + + await WaitUntilDanglingAsync(client, TimeSpan.FromSeconds(30)); + + // Created LAST, i.e. after the targets have expired, so its catch-up meets the dangling link. Its + // query has no `e && e.streamId !== null` guard, which is exactly how a real router projection ends up + // Faulted rather than merely wrong. + var faultyQuery = $"fromStreams(['$et-OrderCreated']).when({{ $any: function(s,e){{ " + + $"linkTo('{FaultyProjection}', e) }} }});"; + await projections.CreateContinuousAsync(FaultyProjection, faultyQuery, trackEmittedStreams: true); + await projections.DisableAsync(FaultyProjection); + await projections.UpdateAsync(FaultyProjection, faultyQuery, emitEnabled: true); + await projections.EnableAsync(FaultyProjection); + } + + private static Task Append(KurrentDBClient c, string stream, StreamState expected, (string Type, string Json) e) => + c.AppendToStreamAsync(stream, expected, + [new EventData(Uuid.NewUuid(), e.Type, Encoding.UTF8.GetBytes(e.Json), Encoding.UTF8.GetBytes("{}"))]); + + // ---------------------------------------------------------------- preconditions + + /// + /// Fails the test unless $et-LineAdded really does hold a link that no longer resolves. Without + /// this, "every link resolves in the rewritten store" is a claim about a source that never had a problem. + /// + public async Task AssertDanglingLinkExistsAsync() + { + await using var client = new KurrentDBClient(KurrentDBClientSettings.Create(ConnectionString)); + var (total, dangling) = await CountLinksAsync(client, "$et-LineAdded"); + _output?.WriteLine($"precondition: $et-LineAdded has {total} link(s), {dangling} dangling"); + dangling.Should().BeGreaterThan(0, + "the whole point of the pure-copy scenario is a source store whose $et-LineAdded holds links to " + + "events that no longer exist — if the fixture did not produce one, the scenario proves nothing"); + } + + private async Task WaitUntilDanglingAsync(KurrentDBClient client, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + var (_, dangling) = await CountLinksAsync(client, "$et-LineAdded"); + if (dangling > 0) return; + await Task.Delay(500); + } + throw new InvalidOperationException( + "The fixture could not produce a dangling link in $et-LineAdded within " + + $"{timeout.TotalSeconds:0}s — the scenarios that depend on it would be meaningless, so the " + + "fixture fails loudly instead of running them."); + } + + /// The status string of a projection on the container under test, or null if absent. + public async Task ProjectionStatusAsync(string name) + { + await using var projections = NewProjections(); + try { return (await projections.GetStatusAsync(name))?.Status; } + catch { return null; } + } + + /// Polls until reports Faulted, or returns the last status seen. + public async Task WaitForFaultedAsync(string name, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + string? last = null; + while (DateTime.UtcNow < deadline) + { + last = await ProjectionStatusAsync(name); + if (last?.Contains("Faulted", StringComparison.OrdinalIgnoreCase) == true) return last; + await Task.Delay(500); + } + return last; + } + + /// Polls until reports Running, or returns the last status seen. + public async Task WaitForRunningAsync(string name, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + string? last = null; + while (DateTime.UtcNow < deadline) + { + last = await ProjectionStatusAsync(name); + if (last?.Contains("Running", StringComparison.OrdinalIgnoreCase) == true) return last; + await Task.Delay(500); + } + return last; + } + + /// Reads a link stream and reports how many links do not resolve to an event. + public static async Task<(int Total, int Dangling)> CountLinksAsync(KurrentDBClient client, string stream) + { + var res = client.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, resolveLinkTos: true); + if (await res.ReadState == ReadState.StreamNotFound) return (0, 0); + int total = 0, dangling = 0; + await foreach (var re in res) + { + total++; + if (re.Event is null) dangling++; + } + return (total, dangling); + } + + private static async Task WaitForLinkCountAsync(KurrentDBClient client, string stream, int atLeast, + TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + long last = -1; + while (DateTime.UtcNow < deadline) + { + var res = client.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, + resolveLinkTos: false); + last = await res.ReadState == ReadState.StreamNotFound ? 0 : await res.LongCountAsync(); + if (last >= atLeast) return; + await Task.Delay(250); + } + // A silent timeout here would let a scenario run against a half-built fixture and then report a + // conclusion about KurrentDB that is really a conclusion about a starved test host. + throw new InvalidOperationException( + $"Fixture setup: '{stream}' only reached {last} of the {atLeast} expected link(s) in " + + $"{timeout.TotalSeconds:0}s."); + } + + // ---------------------------------------------------------------- reading results + + /// + /// Opens a client against the ORIGINAL container. Every assertion in every scenario reads through this — + /// never through the scratch store, which would prove only that the copy engine can write. + /// + public KurrentDBClient NewClient() => new(KurrentDBClientSettings.Create(ConnectionString)); + + /// A projection-management client against the original container. + public KurrentDBProjectionManagementClient NewProjections() => + new(KurrentDBClientSettings.Create(ConnectionString)); + + /// Reads a stream through the original container; empty when it does not exist. + public async Task> ReadAsync(string stream, bool resolveLinkTos = false) + { + await using var client = NewClient(); + var res = client.ReadStreamAsync(Direction.Forwards, stream, StreamPosition.Start, + resolveLinkTos: resolveLinkTos); + if (await res.ReadState == ReadState.StreamNotFound) return []; + var list = new List(); + await foreach (var e in res) list.Add(resolveLinkTos ? e.Event : e.OriginalEvent); + return list; + } + + /// Whether a stream exists in the store right now. + public async Task StreamExistsAsync(string stream) => (await ReadAsync(stream)).Count > 0; + + /// The subdirectory names directly under . + public IReadOnlyList RootEntries() => + Directory.EnumerateDirectories(RootDir).Select(Path.GetFileName).OrderBy(x => x, StringComparer.Ordinal) + .ToList()!; + + /// The container's start time, so a scenario can prove it was never restarted. + public async Task StartedAtAsync() => + (await _docker.Containers.InspectContainerAsync(ContainerId)).State.StartedAt; + + /// Container names currently running that carry the fixture's compose label. + public async Task> ScratchContainersAsync() + { + var all = await _docker.Containers.ListContainersAsync(new ContainersListParameters { All = true }); + return all.SelectMany(c => c.Names ?? []) + .Select(n => n.TrimStart('/')) + .Where(n => n.StartsWith("mp-rewrite-" + ContainerName, StringComparison.Ordinal)) + .ToList(); + } + + /// + /// Starts a second container carrying the SAME compose label, so the sibling guard has a real sibling. + /// + /// + /// It runs the store image with a sleep entrypoint rather than pulling a second image — this host's + /// root filesystem is at 92 %, and the guard cares about the label, not about what the sibling does. + /// + public async Task StartSiblingAsync() + { + var name = $"mp-rewrite-test-sibling-{Guid.NewGuid():N}"[..40]; + var created = await _docker.Containers.CreateContainerAsync(new CreateContainerParameters + { + Image = Image, + Name = name, + Labels = new Dictionary { [DockerStore.ComposeProjectLabel] = ComposeProject }, + Entrypoint = ["/bin/sh", "-c", "sleep 600"] + }); + _extraContainers.Add(created.ID); + await _docker.Containers.StartContainerAsync(created.ID, new ContainerStartParameters()); + return name; + } + + /// The docker client the tool is handed. + public IDockerClient Docker => _docker; + + /// The logger factory the tool is handed. + public ILoggerFactory Loggers => _loggerFactory; + + /// + /// Waits until no client has a gRPC call open. The connected-client guard is about the OPERATOR's clients; + /// a scenario that left its own seeding subscription open would be testing the test. + /// + public async Task WaitUntilNoClientsAsync(TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + var (grpc, _) = await DockerStore.ReadConnectionMetricsAsync(ConnectionString); + if (grpc == 0) return; + await Task.Delay(200); + } + throw new InvalidOperationException( + $"The store still reports open gRPC calls after {timeout.TotalSeconds:0}s; the scenario's own " + + "client would be mistaken for the operator's."); + } + + private static int FreePort() + { + using var l = new System.Net.Sockets.TcpListener(System.Net.IPAddress.Loopback, 0); + l.Start(); + var p = ((System.Net.IPEndPoint)l.LocalEndpoint).Port; + l.Stop(); + return p; + } + + // ---------------------------------------------------------------- teardown + + public async ValueTask DisposeAsync() + { + var store = new DockerStore(_docker, NullLogger.Instance); + foreach (var id in _extraContainers) await Safe(() => store.RemoveAsync(id)); + await Safe(() => store.RemoveAsync(ContainerId)); + foreach (var name in await SafeList()) await Safe(() => store.RemoveAsync(name)); + + // Every directory under the root was filled by a container running as uid 1001; the test process + // cannot delete those files itself (measured: Permission denied on index/stream-existence). + foreach (var dir in Directory.Exists(RootDir) ? Directory.EnumerateDirectories(RootDir).ToList() : []) + await Safe(() => store.PurgeDirectoryAsync(Image, dir)); + await Safe(() => { if (Directory.Exists(RootDir)) Directory.Delete(RootDir, recursive: true); return Task.CompletedTask; }); + + _docker.Dispose(); + _loggerFactory.Dispose(); + } + + private async Task> SafeList() + { + try { return await ScratchContainersAsync(); } + catch { return []; } + } + + private static async Task Safe(Func action) + { + try { await action(); } + catch { /* teardown must never mask the test's own failure */ } + } +} + +/// Routes the tool's logging into xunit's output so a failed scenario shows what the tool did. +internal sealed class TestOutputLoggerProvider(ITestOutputHelper output) : ILoggerProvider +{ + public ILogger CreateLogger(string categoryName) => new XunitLogger(output, categoryName); + public void Dispose() { } + + private sealed class XunitLogger(ITestOutputHelper output, string category) : ILogger + { + public IDisposable BeginScope(TState state) where TState : notnull => NullScope.Instance; + public bool IsEnabled(LogLevel logLevel) => logLevel >= LogLevel.Information; + + public void Log(LogLevel level, EventId id, TState state, Exception? ex, + Func formatter) + { + if (!IsEnabled(level)) return; + try { output.WriteLine($"[{level}] {category}: {formatter(state, ex)}{(ex is null ? "" : " " + ex.Message)}"); } + catch { /* the test has already finished writing */ } + } + + private sealed class NullScope : IDisposable + { + public static readonly NullScope Instance = new(); + public void Dispose() { } + } + } +} diff --git a/src/MicroPlumberd.Rewrite.Tests/RewriteVerificationTests.cs b/src/MicroPlumberd.Rewrite.Tests/RewriteVerificationTests.cs new file mode 100644 index 0000000..ecac2c0 --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/RewriteVerificationTests.cs @@ -0,0 +1,110 @@ +using FluentAssertions; +using MicroPlumberd.Migration; +using MicroPlumberd.Rewrite; +using Xunit; + +namespace MicroPlumberd.Rewrite.Tests; + +/// +/// The verification gate's logic, over real shapes. These are what prove the gate +/// DISCRIMINATES; the end-to-end test proves the abort path around it. +/// +public class RewriteVerificationTests +{ + private static CopyResult Copy(params (string Stream, long Source, long Kept, long Dropped, string Target)[] rows) + { + var r = new CopyResult + { + DryRun = false, + RunTimeUtc = DateTime.UtcNow, + MigrationStats = new Dictionary() + }; + foreach (var (stream, src, kept, dropped, target) in rows) + r.SourceStreams[stream] = new StreamCopyInfo + { + TargetStream = target, SourceCount = src, Kept = kept, Dropped = dropped + }; + return r; + } + + private static RuleReach DropsStream(string name) => + new([s => s == name], [], HasEventLevelDrop: false); + + [Fact] + public void A_faithful_copy_raises_nothing() + { + var copy = Copy(("Order-1", 3, 3, 0, "Order-1"), ("Junk-1", 2, 0, 2, "Junk-1")); + var truth = new Dictionary { ["Order-1"] = 3, ["Junk-1"] = 2 }; + + RewriteVerification.Check(copy, truth, DropsStream("Junk-1")).Should().BeEmpty(); + } + + [Fact] + public void An_engine_that_read_fewer_events_than_the_source_holds_is_caught() + { + // The defect the library's verifier is structurally blind to: the engine's expectation and its result + // fall together, so the destination matches it exactly and RESULT: OK is printed over a partial copy. + var copy = Copy(("Order-1", 2, 2, 0, "Order-1")); + var truth = new Dictionary { ["Order-1"] = 3 }; + + var issues = RewriteVerification.Check(copy, truth, RuleReach.None); + + issues.Should().ContainSingle().Which.Reason.Should().Contain("3").And.Contain("2"); + } + + [Fact] + public void A_stream_the_engine_never_enumerated_at_all_is_caught() + { + var copy = Copy(("Order-1", 3, 3, 0, "Order-1")); + var truth = new Dictionary { ["Order-1"] = 3, ["Forgotten-1"] = 4 }; + + RewriteVerification.Check(copy, truth, RuleReach.None) + .Should().ContainSingle().Which.Subject.Should().Be("Forgotten-1"); + } + + [Fact] + public void A_stream_that_vanished_with_no_rule_to_explain_it_is_caught() + { + // On a pure copy nothing may vanish at all — and this is precisely the case the library's verifier + // reports to the operator as a "fully-dropped source stream (intended)". + var copy = Copy(("Junk-1", 2, 0, 2, "Junk-1")); + var truth = new Dictionary { ["Junk-1"] = 2 }; + + RewriteVerification.Check(copy, truth, RuleReach.None) + .Should().ContainSingle().Which.Reason.Should().Contain("no rule accounts for it"); + } + + [Fact] + public void The_same_vanished_stream_is_accepted_when_a_dropStream_rule_names_it() + { + // The control for the test above: the gate must not simply refuse everything that vanished, or it + // would block every legitimate dropStream run and tell us nothing. + var copy = Copy(("Junk-1", 2, 0, 2, "Junk-1")); + var truth = new Dictionary { ["Junk-1"] = 2 }; + + RewriteVerification.Check(copy, truth, DropsStream("Junk-1")).Should().BeEmpty(); + } + + [Fact] + public void An_event_level_rule_makes_any_disappearance_attributable() + { + // A dropEvent or a transform can empty any stream one event at a time and cannot be asked about a + // specific stream without replaying the copy. Claiming otherwise would produce false refusals, which + // for a tool with no override means blocking a repair. + var copy = Copy(("Order-1", 3, 0, 3, "Order-1")); + var truth = new Dictionary { ["Order-1"] = 3 }; + + RewriteVerification.Check(copy, truth, new RuleReach([], [], HasEventLevelDrop: true)) + .Should().BeEmpty(); + } + + [Fact] + public void A_stream_dropped_under_the_name_a_rename_gave_it_is_still_explained() + { + var copy = Copy(("Old-1", 2, 0, 2, "Old-1")); + var truth = new Dictionary { ["Old-1"] = 2 }; + var rules = new RuleReach([s => s == "New-1"], [("Old-1", "New-1")], HasEventLevelDrop: false); + + RewriteVerification.Check(copy, truth, rules).Should().BeEmpty(); + } +} diff --git a/src/MicroPlumberd.Rewrite.Tests/ScriptHostTests.cs b/src/MicroPlumberd.Rewrite.Tests/ScriptHostTests.cs new file mode 100644 index 0000000..bd3da8a --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/ScriptHostTests.cs @@ -0,0 +1,305 @@ +using System.Diagnostics; +using System.Text.Json.Nodes; +using FluentAssertions; +using MicroPlumberd.Migration; +using MicroPlumberd.Migration.Scripting; +using Xunit; + +namespace MicroPlumberd.Rewrite.Tests; + +/// +/// UT-01..05 — the JavaScript rule host, in-process, no store and no container. +/// +public class ScriptHostTests +{ + // ------------------------------------------------------------------ UT-01 + + /// + /// Every JSON shape a payload can hold has to survive the trip into JavaScript and back. The script here + /// ADDS one field, which forces the real round-trip: the reference-equality fast path (asserted in the + /// test below) would otherwise return the original node and prove nothing about marshalling. + /// + [Fact] + public void UT01_Data_of_every_JSON_kind_survives_the_round_trip_into_the_script() + { + const string payload = """ + {"int":42,"negative":-7,"zero":0,"fraction":1.5,"bool":true,"falsy":false, + "text":"a \"quoted\" \\ back\\slash é ł", + "nothing":null,"empty":{},"emptyArray":[], + "array":[1,"two",null,{"deep":[true,2.25]}], + "nested":{"a":{"b":{"c":"d"}}}} + """; + var input = Ev.Make(data: payload); + var runner = new ScriptRunner("function transform(o){ o.Data.marker = 1; return o; }"); + + var result = runner.Run(input); + + result.Should().NotBeNull(); + var data = result!.Data!.AsObject(); + data["marker"]!.GetValue().Should().Be(1, "the script's own write must be there — otherwise " + + "this test would pass on a host that never ran the script at all"); + + data.Remove("marker"); + // Compare against the source payload re-parsed and re-rendered the same way, so the comparison is + // about VALUES and key order, not about the whitespace the literal above is formatted with. + data.ToJsonString().Should().Be(JsonNode.Parse(payload)!.ToJsonString()); + } + + /// + /// The fidelity guarantee behind a pure copy: a script that does not touch the payload must hand back the + /// SAME node instance, which is how the copy engine knows to write the original bytes byte-for-byte + /// instead of re-rendering the JSON. Break the reference check and this goes red. + /// + [Fact] + public void UT01_An_untouched_payload_comes_back_as_the_very_same_node_so_the_copy_stays_byte_identical() + { + var input = Ev.Make(data: """{"n":1.0,"s":"x"}"""); + var runner = new ScriptRunner("function transform(o){ return o; }"); + + var result = runner.Run(input); + + result.Should().BeSameAs(input, "an identity transform changed nothing, so nothing may be re-serialised"); + } + + /// + /// The known limit of scripting a JSON payload: JavaScript has one number type, so an integer beyond + /// 2^53 cannot survive a script that touches the payload. This test PINS that behaviour rather than + /// claiming a fidelity we do not have — it is the warning in the README, made executable. + /// + [Fact] + public void UT01_An_integer_beyond_2_pow_53_is_rounded_when_the_script_touches_the_payload() + { + var input = Ev.Make(data: """{"id":9007199254740993}"""); + var untouched = new ScriptRunner("function transform(o){ return o; }").Run(input); + var touched = new ScriptRunner("function transform(o){ o.Data.marker = 1; return o; }").Run(input); + + untouched!.Data!.ToJsonString().Should().Contain("9007199254740993", + "an untouched payload never goes through JavaScript's number type"); + touched!.Data!["id"]!.GetValue().Should().Be(9007199254740992L, + "once the script touches the payload the whole object is re-rendered from JS doubles"); + } + + // ------------------------------------------------------------------ UT-02 + + [Theory] + [InlineData("function transform(o){ o.Stream = ''; return o; }")] + [InlineData("function transform(o){ o.EventType = ''; return o; }")] + [InlineData("function transform(o){ return undefined; }")] + public void UT02_An_empty_Stream_or_EventType_drops_the_event_just_like_returning_undefined(string script) + { + new ScriptRunner(script).Run(Ev.Make()).Should().BeNull(); + } + + /// + /// The single most likely authoring mistake in this contract: build a fresh result object and forget a + /// field. Folding an ABSENT property into "" would make that delete every event the rule touched, silently + /// and at exit 0. Only an EXPLICIT empty string means drop. + /// + [Theory] + [InlineData("function transform(o){ return { Data: o.Data, Metadata: o.Metadata }; }", "Stream")] + [InlineData("function transform(o){ return { Stream: o.Stream, Data: o.Data }; }", "EventType")] + [InlineData("function transform(o){ return []; }", "Stream")] + public void UT02_A_returned_object_MISSING_Stream_or_EventType_is_an_error_not_a_silent_drop( + string script, string missing) + { + var runner = new ScriptRunner(script); + + var act = () => runner.Run(Ev.Make(stream: "Order-7", number: 3)); + + act.Should().Throw() + .Where(e => e.Stream == "Order-7" && e.EventNumber == 3) + .WithMessage($"*{missing}*") + .WithMessage("*Order-7#3*", "an operator must be told which event the script mishandled"); + } + + [Fact] + public void UT02_A_Stream_or_EventType_that_is_not_a_string_is_an_error_rather_than_being_coerced() + { + // Same class as the missing property: coercing 123 to "123" would quietly write events into a stream + // nobody named. It cannot be a drop either, so it is the third thing an error must cover. + var runner = new ScriptRunner("function transform(o){ o.Stream = 123; return o; }"); + + var act = () => runner.Run(Ev.Make(stream: "Order-7", number: 3)); + + act.Should().Throw().WithMessage("*Stream*").WithMessage("*string*"); + } + + [Fact] + public void UT02_A_transform_returning_a_non_object_fails_naming_the_event_it_was_processing() + { + var runner = new ScriptRunner("function transform(o){ return 'oops'; }"); + + var act = () => runner.Run(Ev.Make(stream: "Order-7", number: 3)); + + act.Should().Throw() + .Where(e => e.Stream == "Order-7" && e.EventNumber == 3) + .WithMessage("*Order-7#3*") + .WithMessage("*oops*", "an operator must see WHAT the script returned, not just that it was wrong"); + } + + // ------------------------------------------------------------------ UT-03 + + [Fact] + public void UT03_dropStream_is_declared_before_the_per_event_rule_so_a_dropped_stream_never_reaches_update() + { + var runner = new ScriptRunner(""" + dropStream(/^Junk-/); + update("OrderCreated", function(e){ e.Data.seen = true; return e; }); + """); + + runner.Calls.Should().Equal(["DropStream(predicate)", "Transform"], + "the copy engine applies operations in declaration order and short-circuits on a drop, so the " + + "stream drop MUST be registered first for update never to see those events"); + runner.StreamDrops.Should().ContainSingle(); + runner.StreamDrops[0]("Junk-1").Should().BeTrue(); + runner.StreamDrops[0]("Order-1").Should().BeFalse(); + } + + [Fact] + public void UT03_transform_sees_the_event_as_update_left_it() + { + var runner = new ScriptRunner(""" + update("OrderCreated", function(e){ e.Data.stage = "updated"; return e; }); + function transform(o){ o.Data.sawStage = o.Data.stage; return o; } + """); + + var result = runner.Run(Ev.Make(type: "OrderCreated", data: """{"n":1}""")); + + result!.Data!["stage"]!.GetValue().Should().Be("updated"); + result.Data["sawStage"]!.GetValue().Should().Be("updated", + "transform runs last and must observe the update's result, not the original payload"); + } + + [Fact] + public void UT03_update_only_fires_for_its_own_event_type() + { + var runner = new ScriptRunner("""update("OrderCreated", function(e){ e.Data.touched = true; return e; });"""); + + var matched = runner.Run(Ev.Make(type: "OrderCreated")); + var other = runner.Run(Ev.Make(type: "LineAdded")); + + matched!.Data!["touched"].Should().NotBeNull(); + other!.Data!["touched"].Should().BeNull(); + } + + [Fact] + public void UT03_updateById_fires_only_for_the_matching_event_id() + { + var runner = new ScriptRunner( + "updateById(\"" + Ev.KnownId + "\", function(e){ e.Metadata.Note = \"n\"; return e; });"); + + var matched = runner.Run(Ev.Make(id: Ev.KnownId)); + var other = runner.Run(Ev.Make(id: Guid.NewGuid())); + + matched!.Metadata!["Note"]!.GetValue().Should().Be("n"); + matched.Metadata["$correlationId"]!.GetValue().Should().Be("c", + "an update must not disturb the fields it did not name"); + other!.Metadata!["Note"].Should().BeNull(); + } + + [Fact] + public void UT03_dropEvent_can_match_on_the_source_write_date_as_an_ISO_string() + { + var runner = new ScriptRunner("""dropEvent(function(e){ return e.Created < "2026-08-25"; });"""); + + runner.Run(Ev.Make(created: new DateTime(2026, 8, 20, 0, 0, 0, DateTimeKind.Utc))).Should().BeNull(); + runner.Run(Ev.Make(created: new DateTime(2026, 8, 30, 0, 0, 0, DateTimeKind.Utc))).Should().NotBeNull(); + } + + [Fact] + public void UT03_renameType_and_renameStream_are_declared_as_native_operations_not_folded_into_the_transform() + { + var runner = new ScriptRunner(""" + renameStream("Pipeline-old", "Pipeline-new"); + renameType("PipelineStarted", "PipelineStartedV2"); + """); + + runner.Calls.Should().Equal([ + "RenameStream(\"Pipeline-old\"->\"Pipeline-new\")", + "RenameType(\"PipelineStarted\"->\"PipelineStartedV2\")" + ], "a rename needs no per-event script call, so a script with only renames declares no Transform at all"); + } + + // ------------------------------------------------------------------ UT-04 + + [Fact] + public void UT04_A_syntax_error_fails_construction_with_the_line_and_column() + { + var script = "dropStream(\"A\");\nfunction transform(o) { return o; \n"; + + var act = () => new ScriptMigration(script, "rewrite_20260907T101500"); + + var ex = act.Should().Throw().Which; + ex.Line.Should().BeGreaterThan(0); + ex.Column.Should().BeGreaterThan(0); + ex.Message.Should().Contain($"line {ex.Line}").And.Contain($"column {ex.Column}", + "the operator fixes the file by coordinate — 'the script is invalid' is not actionable"); + } + + [Fact] + public void UT04_A_valid_script_yields_an_id_carrying_the_prefix_and_the_script_hash() + { + const string script = "dropStream(\"Junk-1\");"; + var m = new ScriptMigration(script, "rewrite_20260907T101500"); + + m.Id.Should().StartWith("rewrite_20260907T101500_"); + m.Id.Should().Be($"rewrite_20260907T101500_{m.ScriptChecksum[..12]}"); + m.ChecksumOverride.Should().Be(m.ScriptChecksum); + + // Two DIFFERENT scripts must never share a checksum — the history guard is the only thing standing + // between "already applied" and a store rewritten by different rules. + new ScriptMigration("dropStream(\"Junk-2\");", "rewrite_20260907T101500").ScriptChecksum + .Should().NotBe(m.ScriptChecksum); + } + + // ------------------------------------------------------------------ UT-05 + + [Fact] + public void UT05_A_script_that_never_returns_is_aborted_and_the_failure_names_the_event() + { + var runner = new ScriptRunner("function transform(o){ while(true){} }"); + var sw = Stopwatch.StartNew(); + + var act = () => runner.Run(Ev.Make(stream: "Order-9", number: 5)); + + act.Should().Throw() + .Where(e => e.Stream == "Order-9" && e.EventNumber == 5) + .WithMessage("*Order-9#5*") + .WithInnerException(); + sw.Elapsed.Should().BeLessThan(TimeSpan.FromSeconds(30), + "the budget is per event — a runaway script must not hang the whole rewrite"); + } + + /// + /// The budget belongs to the EVENT, not to each entry into JavaScript. Jint resets its own timeout at + /// every call, so three half-budget helpers would sail through; this pins that they do not. + /// + [Fact] + public void UT05_The_time_budget_is_shared_by_every_script_call_made_for_one_event() + { + const string spin = "function(e){ var t = Date.now(); while (Date.now() - t < 900) {} return e; }"; + var script = string.Concat(Enumerable.Repeat("update(\"Slow\", " + spin + ");\n", 3)); + var runner = new ScriptRunner(script); + + var act = () => runner.Run(Ev.Make(type: "Slow", stream: "Slow-1", number: 0)); + + act.Should().Throw() + .WithInnerException("three 0.9 s handlers exceed one 2 s per-event budget, " + + "even though no single call does"); + } + + /// + /// The sandbox: a script cannot reach .NET. If CLR interop were ever switched on, this stops being an + /// error and the tool would hand every operator's script the filesystem. + /// + [Fact] + public void The_script_cannot_reach_the_CLR() + { + var runner = new ScriptRunner("function transform(o){ o.Data.host = System.IO.File.name; return o; }"); + + var act = () => runner.Run(Ev.Make()); + + act.Should().Throw() + .WithMessage("*System*", "there must be no 'System' in the script's world at all"); + } +} diff --git a/src/MicroPlumberd.Rewrite.Tests/ScriptTestHarness.cs b/src/MicroPlumberd.Rewrite.Tests/ScriptTestHarness.cs new file mode 100644 index 0000000..7ef1c3b --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/ScriptTestHarness.cs @@ -0,0 +1,126 @@ +using System.Text.Json.Nodes; +using MicroPlumberd.Migration; +using MicroPlumberd.Migration.Scripting; + +namespace MicroPlumberd.Rewrite.Tests; + +/// +/// Records what a declares, in declaration order, and hands back the generic +/// Transform rule so a test can run one event through the script without a store. +/// +internal sealed class RecordingBuilder : IMigrationBuilder +{ + /// Every operation the migration declared, in order, as Name(args). + public List Calls { get; } = []; + + /// The stream predicates registered by dropStream, in order. + public List> StreamDrops { get; } = []; + + /// The single generic per-event rule, or null when the script declared none. + public Func? Transformer { get; private set; } + + public IMigrationBuilder DropStream(string name) + { + Calls.Add($"DropStream(\"{name}\")"); + StreamDrops.Add(s => s == name); + return this; + } + + public IMigrationBuilder DropStream(Func predicate) + { + Calls.Add("DropStream(predicate)"); + StreamDrops.Add(predicate); + return this; + } + + public IMigrationBuilder DropStreams(params string[] names) + { + Calls.Add($"DropStreams({string.Join(",", names)})"); + return this; + } + + public IMigrationBuilder DropEvent(Func predicate) + { + Calls.Add("DropEvent(predicate)"); + return this; + } + + public IMigrationBuilder RenameType(string oldType, string newType) + { + Calls.Add($"RenameType(\"{oldType}\"->\"{newType}\")"); + return this; + } + + public IMigrationBuilder TransformJson(string type, Action transform) + { + Calls.Add($"TransformJson(\"{type}\")"); + return this; + } + + public IMigrationBuilder RenameStream(string oldName, string newName) + { + Calls.Add($"RenameStream(\"{oldName}\"->\"{newName}\")"); + return this; + } + + public IMigrationBuilder Transform(Func transform) + { + Calls.Add("Transform"); + Transformer = transform; + return this; + } +} + +/// Compiles a script and runs single events through it, in-process, with no store involved. +internal sealed class ScriptRunner +{ + private readonly RecordingBuilder _builder = new(); + + public ScriptRunner(string script, string idPrefix = "ut_20260907T000000") + { + Migration = new ScriptMigration(script, idPrefix); + Migration.Migrate(_builder); + } + + public ScriptMigration Migration { get; } + public IReadOnlyList Calls => _builder.Calls; + public IReadOnlyList> StreamDrops => _builder.StreamDrops; + + /// The generic per-event rule the script compiled to (fails the test if it declared none). + public Func Transformer => + _builder.Transformer ?? throw new InvalidOperationException( + "The script declared no per-event rule — there is nothing to run an event through."); + + public RawEvent? Run(RawEvent e) => Transformer(e); +} + +/// Builds s for the in-process script tests. +internal static class Ev +{ + public static readonly Guid KnownId = Guid.Parse("8d8066f9-bfbb-423e-a990-3fa5f7c6d4d6"); + + public static RawEvent Make(string stream = "Order-1", string type = "OrderCreated", + string data = "{\"n\":1}", string? meta = "{\"$correlationId\":\"c\"}", + ulong number = 0, Guid? id = null, DateTime? created = null) => new() + { + StreamId = stream, + EventNumber = number, + Type = type, + Data = data is null ? null : JsonNode.Parse(data), + Metadata = meta is null ? null : JsonNode.Parse(meta), + EventId = id ?? KnownId, + Created = created ?? new DateTime(2026, 8, 20, 10, 30, 0, DateTimeKind.Utc) + }; + + /// A raw event whose payload is not JSON at all (the binary / corrupt case). + public static RawEvent NonJson(string stream = "Blob-1", string type = "Blob") => new() + { + StreamId = stream, + EventNumber = 0, + Type = type, + Data = null, + Metadata = null, + EventId = KnownId, + Created = new DateTime(2026, 8, 20, 10, 30, 0, DateTimeKind.Utc) + }; +} diff --git a/src/MicroPlumberd.Rewrite.Tests/StoreHealthTests.cs b/src/MicroPlumberd.Rewrite.Tests/StoreHealthTests.cs new file mode 100644 index 0000000..6a739fd --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/StoreHealthTests.cs @@ -0,0 +1,91 @@ +using System.Net; +using FluentAssertions; +using MicroPlumberd.Rewrite; +using Microsoft.Extensions.Logging.Abstractions; +using Xunit; + +namespace MicroPlumberd.Rewrite.Tests; + +/// +/// Readiness. `/health/live` is an HTTP/1.1 endpoint and KurrentDB answers it BEFORE it will serve gRPC on +/// the same port — measured at 528 ms of daylight on a fresh container (2026-09-07, saturn). Every client this +/// tool and this suite open is gRPC, so treating `/health/live` as "ready" opens a window in which the server +/// answers the HTTP/2 preface with GOAWAY HTTP_1_1_REQUIRED. E2E-05 fell into it on the 81-test run, +/// failing in `RewriteFixture.SeedAsync`'s very first append. +/// +public class StoreHealthTests +{ + /// + /// An HTTP/1.1-only server that answers /health/live — exactly the state a starting KurrentDB is in + /// during the window. cannot speak HTTP/2, so gRPC can never succeed against it. + /// + private sealed class HttpOnlyStore : IDisposable + { + private readonly HttpListener _listener = new(); + public int Port { get; } + public string ConnectionString => $"esdb://admin:changeit@127.0.0.1:{Port}?tls=false&tlsVerifyCert=false"; + + public HttpOnlyStore() + { + Port = FreePort(); + _listener.Prefixes.Add($"http://127.0.0.1:{Port}/"); + _listener.Start(); + _ = Task.Run(async () => + { + while (_listener.IsListening) + { + try + { + var ctx = await _listener.GetContextAsync(); + ctx.Response.StatusCode = 204; + ctx.Response.Close(); + } + catch { return; } + } + }); + } + + private static int FreePort() + { + using var l = new System.Net.Sockets.TcpListener(IPAddress.Loopback, 0); + l.Start(); + var p = ((IPEndPoint)l.LocalEndpoint).Port; + l.Stop(); + return p; + } + + public void Dispose() { try { _listener.Stop(); ((IDisposable)_listener).Dispose(); } catch { } } + } + + [Fact] + public async Task Readiness_is_not_reached_while_only_HTTP_1_1_is_being_served() + { + // The store answers /health/live perfectly. If that alone counted as ready, this would return at once + // and every caller would then open a gRPC client against a server that cannot serve one. + using var store = new HttpOnlyStore(); + + var act = () => StoreHealth.WaitLiveAsync(store.ConnectionString, TimeSpan.FromSeconds(6), + NullLogger.Instance); + + (await act.Should().ThrowAsync( + "/health/live is HTTP/1.1 and says nothing about whether gRPC is servable yet")) + .WithMessage("*gRPC*", "the message has to name the thing that was never ready"); + } + + [Fact] + public async Task An_unreachable_store_still_fails_on_the_health_check_itself() + { + // The control: readiness must still fail for the ordinary reason — nothing listening — and say so, + // rather than every failure now being reported as a gRPC problem. + using var l = new System.Net.Sockets.TcpListener(IPAddress.Loopback, 0); + l.Start(); + var port = ((IPEndPoint)l.LocalEndpoint).Port; + l.Stop(); + + var act = () => StoreHealth.WaitLiveAsync( + $"esdb://admin:changeit@127.0.0.1:{port}?tls=false&tlsVerifyCert=false", + TimeSpan.FromSeconds(2), NullLogger.Instance); + + (await act.Should().ThrowAsync()).WithMessage("*health/live*"); + } +} diff --git a/src/MicroPlumberd.Rewrite.Tests/xunit.runner.json b/src/MicroPlumberd.Rewrite.Tests/xunit.runner.json new file mode 100644 index 0000000..e810a97 --- /dev/null +++ b/src/MicroPlumberd.Rewrite.Tests/xunit.runner.json @@ -0,0 +1,5 @@ +{ + "$schema": "https://xunit.net/schema/current/xunit.runner.schema.json", + "parallelizeTestCollections": false, + "parallelizeAssembly": false +} diff --git a/src/MicroPlumberd.Rewrite/DescriptorRecorder.cs b/src/MicroPlumberd.Rewrite/DescriptorRecorder.cs new file mode 100644 index 0000000..34baacc --- /dev/null +++ b/src/MicroPlumberd.Rewrite/DescriptorRecorder.cs @@ -0,0 +1,79 @@ +using System.Text.Json.Nodes; +using MicroPlumberd.Migration; + +namespace MicroPlumberd.Rewrite; + +/// +/// Captures the descriptors a migration compiles to, so the tool can print the rules in its plan and its +/// report BEFORE the operator commits to the run. +/// +/// +/// The engine records the same descriptors in the store's history, but only after the run. An operator being +/// asked to confirm needs to see what the script was understood to mean while there is still time to say no — +/// "it dropped the wrong streams" is not a thing to learn from a history record. +/// +internal sealed class DescriptorRecorder : IMigrationBuilder +{ + private readonly List _descriptors = []; + private readonly List> _streamDrops = []; + private readonly List<(string From, string To)> _streamRenames = []; + private bool _eventLevelDrop; + + public IReadOnlyList Descriptors => _descriptors; + + /// + /// What this rule set can explain about a whole stream disappearing — the real predicates, not their + /// descriptions, because the verification gate has to ASK them about a specific stream name. + /// + public RuleReach Reach => new(_streamDrops, _streamRenames, _eventLevelDrop); + + public IMigrationBuilder DropStream(string name) + { + _streamDrops.Add(s => string.Equals(s, name, StringComparison.Ordinal)); + return Add($"DropStream(\"{name}\")"); + } + + public IMigrationBuilder DropStream(Func predicate) + { + _streamDrops.Add(predicate); + return Add("DropStreamPredicate(script)"); + } + + public IMigrationBuilder DropStreams(params string[] names) + { + var set = new HashSet(names, StringComparer.Ordinal); + _streamDrops.Add(set.Contains); + return Add($"DropStreams({string.Join(",", names)})"); + } + + // DropEvent and Transform can empty a stream one event at a time, so their presence makes ANY stream's + // disappearance attributable. Neither can be asked "would you have emptied THIS stream?" without replaying + // the whole copy, which is why the gate can only be decisive when no such rule exists — most importantly + // for a pure copy, where nothing may vanish at all. + public IMigrationBuilder DropEvent(Func predicate) + { + _eventLevelDrop = true; + return Add("DropEvent(script)"); + } + + public IMigrationBuilder Transform(Func transform) + { + _eventLevelDrop = true; + return Add("Transform(script)"); + } + + public IMigrationBuilder RenameStream(string oldName, string newName) + { + _streamRenames.Add((oldName, newName)); + return Add($"RenameStream(\"{oldName}\"->\"{newName}\")"); + } + + public IMigrationBuilder RenameType(string oldType, string newType) => Add($"RenameType(\"{oldType}\"->\"{newType}\")"); + public IMigrationBuilder TransformJson(string type, Action transform) => Add($"TransformJson(\"{type}\")"); + + private IMigrationBuilder Add(string descriptor) + { + _descriptors.Add(descriptor); + return this; + } +} diff --git a/src/MicroPlumberd.Rewrite/DirectorySwap.cs b/src/MicroPlumberd.Rewrite/DirectorySwap.cs new file mode 100644 index 0000000..8e413a0 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/DirectorySwap.cs @@ -0,0 +1,160 @@ +using Microsoft.Extensions.Logging; + +namespace MicroPlumberd.Rewrite; + +/// +/// The filesystem half of the rewrite: create the new store's directory, swap it in by renaming within one +/// directory, and put the old one back if anything goes wrong. +/// +/// +/// Every rename happens INSIDE , i.e. on one filesystem, so each is a +/// single atomic rename(2). The swap as a whole is two of them and is therefore NOT atomic — the +/// window between them is precisely why exists and why the backup is never deleted. +/// Ownership. The KurrentDB image runs as uid 1001; an operator on a fleet host or the +/// runner user in CI does not. A rename needs write permission on the PARENT directory, not on the +/// directory being renamed, so the swap works regardless of who owns the data. What does not work by itself +/// is the container writing into a directory this tool created, so the new directory is made group/other +/// writable before the scratch store is pointed at it — and it keeps those permissions after the swap, which +/// is what lets the original container keep writing. +/// +public sealed class DirectorySwap(DataLocation data, ILogger logger) +{ + /// Marks a directory this tool created as the in-progress new store. + public const string NewSuffix = ".new."; + + /// Marks the pre-rewrite store, kept forever. + public const string BackupSuffix = ".bak."; + + /// Marks the rewritten store that a rollback displaced. + public const string RolledBackSuffix = ".rolledback."; + + /// A UTC stamp that sorts lexicographically, used to name every directory this tool creates. + public static string Stamp(DateTime utc) => utc.ToString("yyyyMMddTHHmmss"); + + private string Path(string suffix, string stamp) => + System.IO.Path.Combine(data.Parent, data.Name + suffix + stamp); + + /// Creates the empty directory the scratch store will write the rewritten history into. + public string CreateNewStoreDir(string stamp) + { + var dir = Path(NewSuffix, stamp); + Directory.CreateDirectory(dir); + MakeContainerWritable(dir); + logger.LogInformation("Created new-store directory {Dir}.", dir); + return dir; + } + + /// + /// Grants the container's user write access to a directory this tool created. + /// + /// + /// The container runs as a uid the tool has no way to become and often cannot even chown to (that + /// needs root). Widening the mode is the portable option that works for a non-root operator on a fleet + /// host and for the runner user on ubuntu-latest alike. It is only ever applied to a + /// directory this tool just created, never to the operator's existing data. + /// + public static void MakeContainerWritable(string dir) + { + if (OperatingSystem.IsWindows()) return; + File.SetUnixFileMode(dir, + UnixFileMode.UserRead | UnixFileMode.UserWrite | UnixFileMode.UserExecute | + UnixFileMode.GroupRead | UnixFileMode.GroupWrite | UnixFileMode.GroupExecute | + UnixFileMode.OtherRead | UnixFileMode.OtherWrite | UnixFileMode.OtherExecute); + } + + /// + /// Swaps the new store in: data → data.bak.<ts> then data.new.<ts> → data. + /// Returns the backup path. Both containers must already be stopped. + /// + public SwapResult Swap(string newStoreDir, string stamp) + { + var store = data.StoreDir!; + var backup = Path(BackupSuffix, stamp); + + Directory.Move(store, backup); + logger.LogInformation("Renamed {Store} → {Backup}.", store, backup); + try + { + Directory.Move(newStoreDir, store); + } + catch + { + // The half-swapped state is the only one an operator cannot reason about, so it never survives + // this method: put the original back before the exception leaves. + Directory.Move(backup, store); + logger.LogError("Second rename failed; restored {Backup} → {Store}.", backup, store); + throw; + } + logger.LogInformation("Renamed {New} → {Store}.", newStoreDir, store); + return new SwapResult(backup, newStoreDir); + } + + /// Reverses a : puts the rewritten store back under its .new. name and the backup back. + public void Restore(SwapResult swap) + { + var store = data.StoreDir!; + if (Directory.Exists(store) && !Directory.Exists(swap.NewStoreDir)) + Directory.Move(store, swap.NewStoreDir); + if (!Directory.Exists(store) && Directory.Exists(swap.BackupDir)) + Directory.Move(swap.BackupDir, store); + logger.LogWarning("Restored the original store: {Backup} → {Store}.", swap.BackupDir, store); + } + + /// Backup directories present, most recent first. + public IReadOnlyList Backups() => Find(BackupSuffix); + + /// + /// Half-written store directories left by an interrupted run. Root-owned, so the operator cannot remove + /// them by hand — which is exactly why --status has to name them. + /// + public IReadOnlyList NewStoreDirs() => Find(NewSuffix); + + private IReadOnlyList Find(string suffix) + { + if (!Directory.Exists(data.Parent)) return []; + var prefix = data.Name + suffix; + return Directory.EnumerateDirectories(data.Parent) + .Where(d => System.IO.Path.GetFileName(d).StartsWith(prefix, StringComparison.Ordinal)) + .OrderByDescending(d => d, StringComparer.Ordinal) + .ToList(); + } + + /// + /// Swaps a backup back in: the current store becomes data.rolledback.<ts> and the chosen + /// backup becomes data. The container must already be stopped. + /// + public RollbackResult Rollback(string? backupDir, string stamp) + { + var chosen = backupDir ?? Backups().FirstOrDefault() + ?? throw new RewriteRefusedException(ExitCode.FilesystemFailure, + $"No backup to roll back to: nothing matching '{data.Name}{BackupSuffix}*' in {data.Parent}."); + if (!Directory.Exists(chosen)) + throw new RewriteRefusedException(ExitCode.FilesystemFailure, $"Backup not found: {chosen}"); + + var store = data.StoreDir!; + var displaced = Path(RolledBackSuffix, stamp); + Directory.Move(store, displaced); + try + { + Directory.Move(chosen, store); + } + catch + { + Directory.Move(displaced, store); + throw; + } + logger.LogWarning("Rolled back: {Chosen} → {Store}; the rewritten store is kept at {Displaced}.", + chosen, store, displaced); + return new RollbackResult(chosen, displaced); + } +} + +/// Where the swap put things, so it can be undone. +/// The pre-rewrite store. +/// The name the rewritten store had before it became the live one. +public sealed record SwapResult(string BackupDir, string NewStoreDir); + +/// Where a rollback put things. +/// The backup that is now live. +/// Where the rewritten store was moved to. +public sealed record RollbackResult(string RestoredFrom, string DisplacedTo); diff --git a/src/MicroPlumberd.Rewrite/DockerStore.cs b/src/MicroPlumberd.Rewrite/DockerStore.cs new file mode 100644 index 0000000..7f96f4a --- /dev/null +++ b/src/MicroPlumberd.Rewrite/DockerStore.cs @@ -0,0 +1,376 @@ +using System.Globalization; +using Docker.DotNet; +using Docker.DotNet.Models; +using Microsoft.Extensions.Logging; + +namespace MicroPlumberd.Rewrite; + +/// Raised when the tool refuses to proceed. Carries the exit code the process must return. +public sealed class RewriteRefusedException(ExitCode code, string message) : Exception(message) +{ + /// The exit code this refusal maps to. + public ExitCode Code { get; } = code; +} + +/// Where a container keeps its event-store data on the host. +/// Host path of the data directory itself (what is renamed away on a swap). +/// The path the data directory is mounted at INSIDE the container. +/// True for a bind mount (the supported case); false for a named volume. +/// The named volume, when is false. +public sealed record DataLocation(string? StoreDir, string Destination, bool IsBind, string? VolumeName) +{ + /// The directory the data directory lives IN — backups and the new store are created here. + public string Parent => Path.GetDirectoryName(StoreDir + ?? throw new InvalidOperationException("A named volume has no host path."))!; + + /// The data directory's own name (usually data) — backups are named after it. + public string Name => Path.GetFileName(StoreDir!.TrimEnd(Path.DirectorySeparatorChar)); +} + +/// The inspected KurrentDB container the tool operates on. +public sealed record StoreContainer +{ + /// Full docker id. + public required string Id { get; init; } + + /// Container name without the leading slash. + public required string Name { get; init; } + + /// Image reference — the scratch store is started from the same one. + public required string Image { get; init; } + + /// The container's environment, verbatim. + public required IReadOnlyList Env { get; init; } + + /// The com.docker.compose.project label, or null when not compose-managed. + public string? ComposeProject { get; init; } + + /// Where its data lives. + public required DataLocation Data { get; init; } + + /// Whether the container was running when inspected. + public required bool Running { get; init; } + + /// Host port 2113 is published on (null when nothing is published). + public int? PublishedPort { get; init; } + + /// The container's own bridge address — the fallback when no port is published. + public string? BridgeIp { get; init; } +} + +/// +/// Everything the tool does to the docker host and the host filesystem. Kept apart from +/// so the orchestration reads as the seven steps of the design and the docker +/// details are in one place. +/// +public sealed partial class DockerStore(IDockerClient client, ILogger logger) +{ + /// The label compose stamps on every container of a project. + public const string ComposeProjectLabel = "com.docker.compose.project"; + + /// Env keys that name a non-default data directory, most specific first. + private static readonly string[] DbPathEnvKeys = ["KURRENTDB_DB", "EVENTSTORE_DB"]; + + /// Default in-container data directories, newest product name first. + private static readonly string[] DefaultDbPaths = ["/var/lib/kurrentdb", "/var/lib/eventstore"]; + + /// + /// Settings that point at STATE which must move with the data. If one of these resolves outside the + /// directory being swapped, the rewrite cannot be correct and the tool refuses. + /// + /// + /// The index is the dangerous one: the scratch store builds an index for the NEW log, but if the original + /// container keeps its index somewhere this tool does not swap, it comes back on new data with a stale + /// index — silent and catastrophic, and the same shape as the in-memory-database case (D13). Refusing an + /// unsupported layout is the posture the owner already chose for named volumes (ADR 7, decision 7); + /// replicating arbitrary extra mounts is not iteration-1 scope. + /// + private static readonly string[] StatePathEnvKeys = + ["KURRENTDB_DB", "KURRENTDB_INDEX", "EVENTSTORE_DB", "EVENTSTORE_INDEX"]; + + /// + /// Settings that point at a path which is NOT state. Reported, never refused — see + /// . + /// + private static readonly string[] DiagnosticPathEnvKeys = ["KURRENTDB_LOG", "EVENTSTORE_LOG"]; + + /// The metric that counts gRPC calls a client currently has open against the node. + /// + /// Measured on KurrentDB 26.1 (2026-09-07): /stats has proc/tcp/connections, which counts + /// the LEGACY TCP client protocol only and stays 0 for every gRPC client — it cannot serve as this guard. + /// The Prometheus endpoint /metrics exposes kurrentdb_current_incoming_grpc_calls: 0 with no + /// client, 2 with one open $all subscription, back to 0 once the client process exits. + /// kurrentdb_kestrel_connections counts the tool's OWN scrape, so it is reported but never gated on. + /// + public const string OpenGrpcCallsMetric = "kurrentdb_current_incoming_grpc_calls"; + + /// Reported alongside the guard, never gated on — the tool's own HTTP request is one of them. + public const string KestrelConnectionsMetric = "kurrentdb_kestrel_connections"; + + /// The fleet default, used when the operator names no credentials. + /// + /// Kept as a DEFAULT rather than demanded up front: on this fleet it is correct, and a hard fail-fast + /// would tax every 3 a.m. invocation. What makes that safe is that a wrong password now says so + /// ( separates 401/403 from unreachable) instead of sending the + /// operator to diagnose a store that is answering perfectly well. + /// + public const string DefaultUser = "admin"; + + /// + public const string DefaultPassword = "changeit"; + + // ---------------------------------------------------------------- inspect + + /// Inspects the target container. Refuses (exit 3) when docker or the container is not there. + public async Task InspectAsync(string container, CancellationToken ct = default) + { + ContainerInspectResponse r; + try + { + r = await client.Containers.InspectContainerAsync(container, ct).ConfigureAwait(false); + } + catch (DockerContainerNotFoundException) + { + throw new RewriteRefusedException(ExitCode.DockerUnavailable, + $"No such container: '{container}'."); + } + catch (Exception ex) when (ex is HttpRequestException or DockerApiException or IOException) + { + throw new RewriteRefusedException(ExitCode.DockerUnavailable, + $"Cannot reach the docker daemon: {ex.Message}"); + } + + var env = (r.Config?.Env ?? []).ToList(); + var data = ResolveDataLocation(env, r.Mounts ?? []); + + var labels = r.Config?.Labels; + string? project = null; + labels?.TryGetValue(ComposeProjectLabel, out project); + + return new StoreContainer + { + Id = r.ID, + Name = r.Name.TrimStart('/'), + Image = r.Config?.Image ?? r.Image, + Env = env, + ComposeProject = string.IsNullOrWhiteSpace(project) ? null : project, + Data = data, + Running = r.State?.Running ?? false, + PublishedPort = ReadPublishedPort(r), + BridgeIp = ReadBridgeIp(r) + }; + } + + /// + /// Picks the mount that holds the store: the one whose destination is the env-configured DB path when one + /// is set, otherwise the first default path present. UT-08. + /// + /// + /// The env setting WINS over the defaults, and deliberately so: a container configured with + /// KURRENTDB_DB=/data that also happens to mount something at /var/lib/kurrentdb would + /// otherwise have the wrong directory renamed away — the one failure in this tool that cannot be undone + /// by looking at a backup, because the backup would be of the wrong thing. + /// + public static DataLocation ResolveDataLocation(IReadOnlyList env, IList mounts) + { + var configured = DbPathEnvKeys.Select(k => EnvValue(env, k)).FirstOrDefault(v => v is not null); + var candidates = configured is not null ? new[] { configured } : DefaultDbPaths; + + foreach (var dest in candidates) + { + var m = mounts.FirstOrDefault(x => PathEquals(x.Destination, dest)); + if (m is null) continue; + var isBind = string.Equals(m.Type, "bind", StringComparison.OrdinalIgnoreCase); + return new DataLocation(isBind ? m.Source : null, dest, isBind, isBind ? null : m.Name); + } + + throw new RewriteRefusedException(ExitCode.GuardRefusal, + $"The container has no mount at its data directory ({string.Join(" or ", candidates)}), so its " + + "store lives inside the container's writable layer and cannot be swapped. Mount the data " + + "directory (a bind mount is what this tool supports) before rewriting it."); + } + + /// + /// Refuses when a path-valued STATE setting resolves outside the directory this tool swaps, and reports + /// (without refusing) a log path that does. + /// + /// + /// Deviation, deliberate and flagged: the review list names KURRENTDB_LOG alongside + /// _DB/_INDEX. Logs are not state — nothing about a log path outside the mount makes a + /// rewrite incorrect, and KurrentDB's own default (/var/log/kurrentdb) IS outside the data + /// directory. Refusing on it would block a legitimate 3 a.m. repair, with no override, for no safety gain + /// — the same "false refusal blocks the repair" hazard this tool is otherwise careful about. So a log path + /// is surfaced in the report and never gates the run. + /// + public static (IReadOnlyList Refusals, IReadOnlyList Notes) CheckStatePathsInsideMount( + IReadOnlyList env, DataLocation data) + { + var refusals = new List(); + var notes = new List(); + + foreach (var key in StatePathEnvKeys) + { + var value = EnvValue(env, key); + if (value is null || IsInside(value, data.Destination)) continue; + refusals.Add($"{key}={value} resolves outside the directory this tool swaps " + + $"({data.Destination}), so the container would come back on the new log with state " + + $"this rewrite never touched. Move it inside {data.Destination} and try again."); + } + + foreach (var key in DiagnosticPathEnvKeys) + { + var value = EnvValue(env, key); + if (value is null || IsInside(value, data.Destination)) continue; + notes.Add($"note : {key}={value} is outside {data.Destination}; it is not state, so it is " + + "not swapped and not a reason to refuse"); + } + + return (refusals, notes); + } + + /// Whether a container path is the mount root or sits beneath it. + internal static bool IsInside(string path, string root) + { + var p = path.TrimEnd('/'); + var r = root.TrimEnd('/'); + return string.Equals(p, r, StringComparison.Ordinal) + || p.StartsWith(r + "/", StringComparison.Ordinal); + } + + /// Reads a KEY=value entry out of a container's environment. + public static string? EnvValue(IReadOnlyList env, string key) + { + foreach (var e in env) + { + var i = e.IndexOf('='); + if (i > 0 && e.AsSpan(0, i).SequenceEqual(key)) return e[(i + 1)..]; + } + return null; + } + + private static bool PathEquals(string? a, string b) => + a is not null && a.TrimEnd('/') == b.TrimEnd('/'); + + private static int? ReadPublishedPort(ContainerInspectResponse r) + { + var bindings = r.NetworkSettings?.Ports; + if (bindings is null) return null; + foreach (var (port, list) in bindings) + { + if (!port.StartsWith("2113/", StringComparison.Ordinal) || list is null) continue; + foreach (var b in list) + if (int.TryParse(b.HostPort, NumberStyles.Integer, CultureInfo.InvariantCulture, out var p) && p > 0) + return p; + } + return null; + } + + private static string? ReadBridgeIp(ContainerInspectResponse r) + { + var networks = r.NetworkSettings?.Networks; + if (networks is not null) + foreach (var n in networks.Values) + if (!string.IsNullOrWhiteSpace(n.IPAddress)) return n.IPAddress; + return string.IsNullOrWhiteSpace(r.NetworkSettings?.IPAddress) ? null : r.NetworkSettings.IPAddress; + } + + /// + /// The address the tool reads the OLD store through: its published 2113 port when there is one, else the + /// container's own bridge address. + /// + /// + /// The fallback matters on the fleet, where compose files publish nothing and the store is reachable only + /// on the bridge network. It works because the tool runs on the docker host itself; from anywhere else the + /// bridge address is unroutable and the operator must publish the port. + /// + public static string ConnectionString(StoreContainer c, string user = DefaultUser, string password = DefaultPassword) + { + var credentials = $"{Uri.EscapeDataString(user)}:{Uri.EscapeDataString(password)}"; + if (c.PublishedPort is { } port) + return $"esdb://{credentials}@127.0.0.1:{port}?tls=false&tlsVerifyCert=false"; + if (c.BridgeIp is { } ip) + return $"esdb://{credentials}@{ip}:2113?tls=false&tlsVerifyCert=false"; + throw new RewriteRefusedException(ExitCode.GuardRefusal, + $"Container '{c.Name}' publishes no port for 2113 and has no bridge address — the tool cannot " + + "read its store. Publish 2113 on the host and try again."); + } + + // ---------------------------------------------------------------- guards + + /// Running containers of the same compose project, excluding the target itself. + public async Task> FindRunningSiblingsAsync(StoreContainer c, CancellationToken ct = default) + { + if (c.ComposeProject is null) return []; + + var all = await client.Containers.ListContainersAsync(new ContainersListParameters { All = false }, ct) + .ConfigureAwait(false); + return all + .Where(x => x.ID != c.Id) + .Where(x => x.Labels is not null + && x.Labels.TryGetValue(ComposeProjectLabel, out var p) + && string.Equals(p, c.ComposeProject, StringComparison.Ordinal)) + .Select(x => x.Names?.FirstOrDefault()?.TrimStart('/') ?? x.ID[..12]) + .OrderBy(n => n, StringComparer.Ordinal) + .ToList(); + } + + /// + /// gRPC calls a client currently has open against the store, and the raw kestrel connection count. + /// Reads /metrics over plain HTTP — which is NOT a gRPC call, so the tool never counts itself. + /// + public static async Task<(int OpenGrpcCalls, int KestrelConnections)> ReadConnectionMetricsAsync( + string connectionString, CancellationToken ct = default) + { + var baseUri = StoreHttp.BaseUriOf(connectionString); + string text; + try + { + using var resp = await StoreHttp.GetAsync(connectionString, "metrics", ct).ConfigureAwait(false); + // A rejected LOGIN is a different problem from a store that is down, and telling them apart is the + // difference between "check your password" and an hour spent diagnosing a healthy store. + if (resp.StatusCode is System.Net.HttpStatusCode.Unauthorized or System.Net.HttpStatusCode.Forbidden) + throw new RewriteRefusedException(ExitCode.GuardRefusal, + $"The store at {baseUri} rejected the credentials ({(int)resp.StatusCode} " + + $"{resp.StatusCode}). It is running and reachable — the user or password is wrong. Pass " + + "--user/--password, or set MP_REWRITE_USER / MP_REWRITE_PASSWORD."); + resp.EnsureSuccessStatusCode(); + text = await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + } + catch (Exception ex) when (ex is HttpRequestException or TaskCanceledException) + { + throw new RewriteRefusedException(ExitCode.DockerUnavailable, + $"The store at {baseUri} did not answer /metrics ({ex.Message}). The connected-client guard " + + "cannot be evaluated, and running blind would risk losing a live writer's data."); + } + return (ReadMetric(text, OpenGrpcCallsMetric), ReadMetric(text, KestrelConnectionsMetric)); + } + + /// Reads a single-sample Prometheus gauge out of a /metrics body. + /// + /// Returns -1 when the metric is ABSENT, which the caller must treat as "cannot tell" rather than "zero". + /// A future KurrentDB that renames the counter would otherwise silently turn this guard off — the exact + /// shape of failure the guard exists to prevent. + /// + internal static int ReadMetric(string metricsBody, string name) + { + foreach (var line in metricsBody.Split('\n')) + { + var l = line.AsSpan().Trim(); + if (!l.StartsWith(name)) continue; + var rest = l[name.Length..]; + if (rest.Length > 0 && rest[0] == '{') + { + var close = rest.IndexOf('}'); + if (close < 0) continue; + rest = rest[(close + 1)..]; + } + else if (rest.Length > 0 && rest[0] != ' ') continue; // a longer metric name that merely starts the same + + var parts = rest.Trim().ToString().Split(' ', StringSplitOptions.RemoveEmptyEntries); + if (parts.Length > 0 && double.TryParse(parts[0], NumberStyles.Float, CultureInfo.InvariantCulture, + out var v)) + return (int)v; + } + return -1; + } +} diff --git a/src/MicroPlumberd.Rewrite/DockerStoreLifecycle.cs b/src/MicroPlumberd.Rewrite/DockerStoreLifecycle.cs new file mode 100644 index 0000000..64e8eb0 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/DockerStoreLifecycle.cs @@ -0,0 +1,273 @@ +using Docker.DotNet; +using Docker.DotNet.Models; +using Microsoft.Extensions.Logging; + +namespace MicroPlumberd.Rewrite; + +public sealed partial class DockerStore +{ + /// Pulls when it is not present locally. Refuses with exit 3 if it cannot. + public async Task EnsureImageAsync(string image, CancellationToken ct = default) + { + if (await ImageExistsAsync(image, ct).ConfigureAwait(false)) return; + + var (repository, tag) = SplitTag(image); + logger.LogInformation("Pulling image {Image}…", image); + try + { + await client.Images.CreateImageAsync(new ImagesCreateParameters { FromImage = repository, Tag = tag }, + null, new Progress(), ct).ConfigureAwait(false); + } + catch (Exception ex) when (ex is DockerApiException or HttpRequestException or IOException) + { + throw new RewriteRefusedException(ExitCode.DockerUnavailable, + $"Could not pull image '{image}': {ex.Message}"); + } + + // A pull that quietly resolves nothing would otherwise surface as an opaque container-create failure. + if (!await ImageExistsAsync(image, ct).ConfigureAwait(false)) + throw new RewriteRefusedException(ExitCode.DockerUnavailable, + $"Image '{image}' is still not present after pulling it."); + } + + private async Task ImageExistsAsync(string image, CancellationToken ct) + { + try + { + await client.Images.InspectImageAsync(image, ct).ConfigureAwait(false); + return true; + } + catch (DockerImageNotFoundException) { return false; } + } + + private static (string Repository, string Tag) SplitTag(string image) + { + var colon = image.LastIndexOf(':'); + var slash = image.LastIndexOf('/'); + return colon > slash && colon >= 0 ? (image[..colon], image[(colon + 1)..]) : (image, "latest"); + } + + /// Stops a container and waits for it to actually be stopped. + public async Task StopAsync(string id, CancellationToken ct = default) + { + await client.Containers.StopContainerAsync(id, new ContainerStopParameters { WaitBeforeKillSeconds = 30 }, ct) + .ConfigureAwait(false); + logger.LogInformation("Stopped container {Id}.", Short(id)); + } + + /// Starts a container. + public async Task StartAsync(string id, CancellationToken ct = default) + { + await client.Containers.StartContainerAsync(id, new ContainerStartParameters(), ct).ConfigureAwait(false); + logger.LogInformation("Started container {Id}.", Short(id)); + } + + /// Force-removes a container, ignoring the case where it is already gone. + public async Task RemoveAsync(string id, CancellationToken ct = default) + { + try + { + await client.Containers.RemoveContainerAsync(id, new ContainerRemoveParameters { Force = true }, ct) + .ConfigureAwait(false); + logger.LogInformation("Removed container {Id}.", Short(id)); + } + catch (DockerContainerNotFoundException) { /* already gone — the desired state */ } + } + + /// Whether the container is currently running. + public async Task IsRunningAsync(string id, CancellationToken ct = default) + { + try + { + var r = await client.Containers.InspectContainerAsync(id, ct).ConfigureAwait(false); + return r.State?.Running ?? false; + } + catch (DockerContainerNotFoundException) { return false; } + } + + /// + /// Creates and starts the scratch store: the SAME image as the old container, its environment plus the + /// three settings the copy engine requires, bound to , on a random host port + /// published to loopback only. + /// + /// + /// The scratch store is published on 127.0.0.1 rather than 0.0.0.0 because for the minutes it + /// exists it holds a full copy of production history with authentication effectively disabled + /// (KURRENTDB_INSECURE=true, which the copy engine needs). It has no business being reachable from + /// the network. + /// + public async Task StartScratchAsync(StoreContainer old, string hostDataDir, string name, + CancellationToken ct = default) + { + var env = BuildScratchEnv(old.Env); + var port = FreeTcpPort(); + + var created = await client.Containers.CreateContainerAsync(new CreateContainerParameters + { + Image = old.Image, + Name = name, + Env = env, + ExposedPorts = new Dictionary { ["2113"] = default }, + HostConfig = new HostConfig + { + // Bind, never a new network: creating one on this host hands out a subnet that collides with + // the company VPN. The default bridge plus a loopback-published port is all the tool needs. + Binds = [$"{hostDataDir}:{old.Data.Destination}"], + PortBindings = new Dictionary> + { + ["2113"] = [new PortBinding { HostIP = "127.0.0.1", HostPort = port.ToString() }] + }, + AutoRemove = false + } + }, ct).ConfigureAwait(false); + + await client.Containers.StartContainerAsync(created.ID, new ContainerStartParameters(), ct) + .ConfigureAwait(false); + logger.LogInformation("Scratch store {Name} started on 127.0.0.1:{Port} over {Dir}.", name, port, hostDataDir); + + return new ScratchStore(created.ID, name, port, + $"esdb://admin:changeit@127.0.0.1:{port}?tls=false&tlsVerifyCert=false"); + } + + /// + /// The old container's environment with the copy engine's three requirements forced on, replacing any + /// existing value rather than appending a second one (docker keeps the LAST, but a duplicated key in an + /// inspect output is exactly the kind of thing that makes an incident harder to read). + /// + internal static List BuildScratchEnv(IReadOnlyList oldEnv) + { + var forced = new Dictionary(StringComparer.Ordinal) + { + ["KURRENTDB_RUN_PROJECTIONS"] = "All", + ["KURRENTDB_START_STANDARD_PROJECTIONS"] = "true", + ["KURRENTDB_INSECURE"] = "true", + // The scratch store must be a REAL store on the bind-mounted directory. An old container running + // with an in-memory database would otherwise hand the scratch store the same setting and the copy + // would be written to nothing. + ["KURRENTDB_MEM_DB"] = "false" + }; + + var result = new List(); + foreach (var e in oldEnv) + { + var i = e.IndexOf('='); + var key = i > 0 ? e[..i] : e; + if (!forced.ContainsKey(key)) result.Add(e); + } + result.AddRange(forced.Select(kv => $"{kv.Key}={kv.Value}")); + return result; + } + + private static int FreeTcpPort() + { + using var l = new System.Net.Sockets.TcpListener(System.Net.IPAddress.Loopback, 0); + l.Start(); + var port = ((System.Net.IPEndPoint)l.LocalEndpoint).Port; + l.Stop(); + return port; + } + + /// + /// Scratch containers this tool has running or left behind for , so + /// --status can tell an operator that a rewrite is in flight — or was interrupted. + /// + public async Task> FindScratchContainersAsync(string containerName, + CancellationToken ct = default) + { + var all = await client.Containers.ListContainersAsync(new ContainersListParameters { All = true }, ct) + .ConfigureAwait(false); + var prefix = $"mp-rewrite-{containerName}-"; + return all.SelectMany(x => x.Names ?? []) + .Select(n => n.TrimStart('/')) + .Where(n => n.StartsWith(prefix, StringComparison.Ordinal)) + .OrderBy(n => n, StringComparer.Ordinal) + .ToList(); + } + + private static string Short(string id) => id.Length > 12 ? id[..12] : id; +} + +/// A started scratch store: the container the rewrite is copied INTO before the swap. +/// Docker container id. +/// Container name (mp-rewrite-<ctr>-<ts>). +/// Host port 2113 is published on, bound to loopback. +/// KurrentDB connection string for the scratch store. +public sealed record ScratchStore(string Id, string Name, int Port, string ConnectionString); + +public sealed partial class DockerStore +{ + /// + /// Deletes a directory the STORE CONTAINER wrote into, from inside a throwaway root container. + /// + /// + /// Measured on saturn (2026-09-07), KurrentDB 26.1: the image runs as uid 1001 and creates + /// index/stream-existence/ mode 0755 owned by 1001. An operator running the tool as uid 1000 gets + /// Permission denied deleting the files inside it, so an ordinary recursive delete of + /// data.new.<ts> — which a dry run and every pre-swap failure must do — cannot work. Widening + /// the mode of the directory the tool creates does not help either: the container's own subdirectories are + /// created with the container's umask, not the parent's. + /// So the delete is done by the only user who can: root inside a container from the same image, with + /// the PARENT directory bind-mounted. The tool never needs root on the host. + /// + public async Task PurgeDirectoryAsync(string image, string dir, CancellationToken ct = default) + { + if (!Directory.Exists(dir)) return; + + var full = Path.GetFullPath(dir).TrimEnd(Path.DirectorySeparatorChar); + if (Path.GetDirectoryName(full) is null) + throw new ArgumentException($"Refusing to purge a filesystem root: '{dir}'.", nameof(dir)); + + // Try the cheap path first — an empty or tool-owned directory needs no container at all. + try + { + Directory.Delete(full, recursive: true); + logger.LogInformation("Removed {Dir}.", full); + return; + } + catch (Exception ex) when (ex is UnauthorizedAccessException or IOException) + { + logger.LogInformation("{Dir} holds files written by the container's user; removing its contents " + + "from a root helper container.", full); + } + + // THE INTERLOCK IS THE MOUNT. The container gets ONLY the directory being deleted, and deletes its + // CONTENTS — so the blast radius equals that directory by construction, for every caller, and no + // convention about names or path depth has to hold for that to be true. The parent (which holds the + // LIVE store and every backup) is never mounted into a root container at all; the now-empty directory + // is removed from the host afterwards, which needs no privileges. + var created = await client.Containers.CreateContainerAsync(new CreateContainerParameters + { + Image = image, + Name = $"mp-rewrite-purge-{Guid.NewGuid():N}"[..40], + User = "0:0", + Entrypoint = ["/bin/sh", "-c", "rm -rf /purge/..?* /purge/.[!.]* /purge/*"], + HostConfig = new HostConfig { Binds = [$"{full}:/purge"], AutoRemove = false } + }, ct).ConfigureAwait(false); + + try + { + await client.Containers.StartContainerAsync(created.ID, new ContainerStartParameters(), ct) + .ConfigureAwait(false); + var wait = await client.Containers.WaitContainerAsync(created.ID, ct).ConfigureAwait(false); + if (wait.StatusCode != 0) + throw new RewriteRefusedException(ExitCode.FilesystemFailure, + $"Could not remove '{full}': the helper container exited with {wait.StatusCode}."); + } + finally + { + await RemoveAsync(created.ID, ct).ConfigureAwait(false); + } + + // The helper emptied it; removing an empty directory needs no privileges. + try + { + Directory.Delete(full, recursive: true); + } + catch (Exception ex) when (ex is UnauthorizedAccessException or IOException) + { + throw new RewriteRefusedException(ExitCode.FilesystemFailure, + $"'{full}' could not be removed even after its contents were purged: {ex.Message}"); + } + logger.LogInformation("Removed {Dir} (contents via a root helper container).", full); + } +} diff --git a/src/MicroPlumberd.Rewrite/ExitCode.cs b/src/MicroPlumberd.Rewrite/ExitCode.cs new file mode 100644 index 0000000..2555f58 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/ExitCode.cs @@ -0,0 +1,27 @@ +namespace MicroPlumberd.Rewrite; + +/// +/// The process exit codes. They are part of the tool's contract — an operator's shell script and CI branch on +/// them — so each one names a DIFFERENT recovery, and the store's state for each is stated here and in the +/// README. +/// +public enum ExitCode +{ + /// The rewrite completed and was verified; the container runs on the new data. + Ok = 0, + + /// A guard refused: a sibling container is running, or a client is connected. Nothing changed. + GuardRefusal = 1, + + /// The script does not parse. Nothing was started; nothing changed. + ScriptError = 2, + + /// Docker is unreachable, the container does not exist, or the image could not be pulled. Nothing changed. + DockerUnavailable = 3, + + /// The copy engine or the verification failed. The scratch store is removed; the OLD store is untouched. + EngineFailure = 4, + + /// The filesystem swap failed. The original data directory has been restored and the container started. + FilesystemFailure = 5 +} diff --git a/src/MicroPlumberd.Rewrite/MicroPlumberd.Rewrite.csproj b/src/MicroPlumberd.Rewrite/MicroPlumberd.Rewrite.csproj new file mode 100644 index 0000000..4ca26c6 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/MicroPlumberd.Rewrite.csproj @@ -0,0 +1,52 @@ + + + + Exe + net10.0 + enable + enable + mp-rewrite + MicroPlumberd.Rewrite + embedded + true + + true + mp-rewrite + MicroPlumberd.Rewrite + mp-rewrite — rewrites the history of a KurrentDB running in docker into a NEW store and swaps it in place, optionally through a JavaScript rule script. The old data directory is kept as a backup and never edited. + mp-rewrite + Rafal Maciag + Rafal Maciag + logo-squere.png + README.md + MIT + https://github.com/modelingevolution/micro-plumberd + https://modelingevolution.github.io/micro-plumberd/ + EventStore;KurrentDB;CQRS;EventSourcing;Migration;dotnet-tool + + true + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/src/MicroPlumberd.Rewrite/Program.cs b/src/MicroPlumberd.Rewrite/Program.cs new file mode 100644 index 0000000..987839e --- /dev/null +++ b/src/MicroPlumberd.Rewrite/Program.cs @@ -0,0 +1,109 @@ +using Docker.DotNet; +using MicroPlumberd.Rewrite; +using Microsoft.Extensions.Logging; + +// Program parses arguments and NOTHING else — every behaviour lives in RewriteCommand, which the end-to-end +// suite calls in-process, so the tested code and the shipped code are the same code. + +const string Usage = """ +mp-rewrite [--script ] [--eval ""] [--dry-run] + [--no-projection-copy] [--yes] [--user ] [--password

] +mp-rewrite --rollback [] +mp-rewrite --status + +Exit codes: 0 ok · 1 guard refusal · 2 script error · 3 docker unreachable/pull failure + 4 engine or verification failure (old store untouched) · 5 filesystem/swap failure (restored) +"""; + +if (args.Length == 0 || args[0] is "-h" or "--help") +{ + Console.WriteLine(Usage); + return args.Length == 0 ? (int)ExitCode.GuardRefusal : (int)ExitCode.Ok; +} + +RewriteOptions options; +try +{ + options = ParseArgs(args); +} +catch (ArgumentException ex) +{ + Console.Error.WriteLine(ex.Message); + Console.Error.WriteLine(); + Console.Error.WriteLine(Usage); + return (int)ExitCode.GuardRefusal; +} + +using var loggerFactory = LoggerFactory.Create(b => b + .SetMinimumLevel(LogLevel.Information) + .AddSimpleConsole(o => { o.SingleLine = true; o.TimestampFormat = "HH:mm:ss "; })); + +using var docker = new DockerClientConfiguration().CreateClient(); +var report = await RewriteCommand.RunAsync(options, docker, loggerFactory); +Console.WriteLine(report.Format()); +return (int)report.Code; + +static RewriteOptions ParseArgs(string[] args) +{ + var container = args[0]; + if (container.StartsWith('-')) + throw new ArgumentException("The first argument must be the container id or name."); + + string? script = null, eval = null, backupDir = null; + // Credentials: flag beats environment beats the fleet default. + var user = Environment.GetEnvironmentVariable("MP_REWRITE_USER") ?? DockerStore.DefaultUser; + var password = Environment.GetEnvironmentVariable("MP_REWRITE_PASSWORD") ?? DockerStore.DefaultPassword; + bool dryRun = false, noProjectionCopy = false, yes = false, forceVolume = false; + var mode = RewriteMode.Rewrite; + + for (var i = 1; i < args.Length; i++) + { + switch (args[i]) + { + case "--script": script = Next(args, ref i, "--script"); break; + case "--eval": eval = Next(args, ref i, "--eval"); break; + case "--user": user = Next(args, ref i, "--user"); break; + case "--password": password = Next(args, ref i, "--password"); break; + case "--dry-run": dryRun = true; break; + case "--no-projection-copy": noProjectionCopy = true; break; + case "--yes" or "-y": yes = true; break; + // Still accepted so it fails with its REASON rather than "unknown argument"; RewriteCommand + // refuses it. Deliberately absent from the usage above — it is reserved, not offered. + case "--force-volume-copy": forceVolume = true; break; + case "--status": mode = RewriteMode.Status; break; + case "--rollback": + mode = RewriteMode.Rollback; + // The backup directory is optional; only consume the next token when it is not a flag. + if (i + 1 < args.Length && !args[i + 1].StartsWith('-')) backupDir = args[++i]; + break; + default: throw new ArgumentException($"Unknown argument: {args[i]}"); + } + } + + // Deliberately NOT checked here: --script with --eval is a script-input error, and RewriteOptions + // already rejects it. Deciding it in two places is how one invalid command line ends up with two + // different exit codes depending on which check happens to run first. + return new RewriteOptions + { + Container = container, + ScriptPath = script, + Eval = eval, + DryRun = dryRun, + NoProjectionCopy = noProjectionCopy, + Yes = yes, + ForceVolumeCopy = forceVolume, + Mode = mode, + BackupDir = backupDir, + User = user, + Password = password, + // The fault injection is a TEST hook and is deliberately reachable only through an environment + // variable, never a command-line flag: nothing an operator can mistype should be able to arm it. + FailAfterSwapForTest = Environment.GetEnvironmentVariable("MP_REWRITE_TEST_FAIL_AFTER_SWAP") == "1" + }; +} + +static string Next(string[] args, ref int i, string flag) +{ + if (i + 1 >= args.Length) throw new ArgumentException($"{flag} needs a value."); + return args[++i]; +} diff --git a/src/MicroPlumberd.Rewrite/README.md b/src/MicroPlumberd.Rewrite/README.md new file mode 100644 index 0000000..3f9e113 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/README.md @@ -0,0 +1,189 @@ +# `mp-rewrite` + +Rewrites the history of a **KurrentDB running in docker** into a **new store**, then puts that store in place +of the old one. It never edits the old store: the previous data directory is renamed to `data.bak.` +and is never deleted by the tool. + +You give it the id or name of the container, and optionally a JavaScript rule script. + +``` +mp-rewrite [--script ] [--eval ""] [--dry-run] + [--no-projection-copy] [--yes] +mp-rewrite --rollback [] +mp-rewrite --status +``` + +| Option | Meaning | +|---|---| +| `--script ` | Rule script to run every event through. | +| `--user` / `--password` | Credentials for the store being rewritten (or `MP_REWRITE_USER` / `MP_REWRITE_PASSWORD`). Default `admin`/`changeit`. | +| `--eval ""` | The same, inline. Mutually exclusive with `--script`. | +| `--dry-run` | Copy nothing. Print the per-rule counts and the names of every affected stream. | +| `--no-projection-copy` | Do not pre-create the app's user projections on the new store (default: copy them). | +| `--yes` | Skip the confirmation prompt. | +| `--rollback [dir]` | Swap the most recent backup (or the given one) back and restart the container. | +| `--status` | Print the data location, the backups present, and whether a rewrite is safe right now. | + +**With no script the tool performs a pure copy — and that alone repairs a store.** System streams, `$et-*` +streams and `$>` link events are never copied, so a store whose command-router projection is Faulted on a +dangling link comes back with the links rebuilt from the events that actually exist. That is the case this +tool was written for. + +## Installing + +```bash +dotnet tool install -g MicroPlumberd.Rewrite # needs the .NET SDK +``` + +Devices without an SDK (neurons) use the single-file build attached to each release — one file, nothing to +install: + +```bash +curl -L -o mp-rewrite /mp-rewrite--linux-arm64 +chmod +x mp-rewrite && ./mp-rewrite --status +``` + +## What a run does + +1. **Parses the script** — before anything else, so a typo costs nothing. +2. **Inspects the container**: image, environment, compose label, and which mount holds the store. The data + directory is the mount whose destination matches `KURRENTDB_DB` / `EVENTSTORE_DB` when either is set, + otherwise `/var/lib/kurrentdb` or `/var/lib/eventstore`. +3. **Guards** (below). These run before the tool opens its own connection to the store. +4. **Starts a scratch store** from the same image on an empty `data.new.` beside the current one. +5. **Copies every event** in `$all` commit order through the rules, and pre-creates the app's user projections + so the merge streams are rebuilt in commit order. +6. **Verifies** the destination against what the copy believes it wrote — per-stream counts and a write + checksum. A mismatch aborts *before* the swap. +7. **Swaps**: stop both containers, `data → data.bak.`, `data.new. → data`, start the original + container, wait for `/health/live`, check that no projection is Faulted. + +A real run appends a `MigrationApplied` record to the new store's `mp-migrations` stream — the script's +sha256, the rules it compiled to, and the counts — so the history of rewrites travels with the data. + +## Refusals + +The tool **refuses rather than warns**, and these two checks have no `--force`: if either is true, a writer +could lose data. + +- **A sibling container of the same compose project is running.** Every container carrying the target's + `com.docker.compose.project` label must be stopped first. The message names them. +- **A client has work in flight against the store.** + +### The connected-client check, and exactly what it reads + +`/stats` **cannot** answer this on KurrentDB 26.1. Its only connection counter is `proc.tcp.connections`, +which counts the **legacy TCP client protocol** — measured at 0 with a live gRPC `$all` subscription open. + +The guard therefore reads the Prometheus endpoint `/metrics`: + +| Metric | Used | Measured: no client / open `$all` subscription / client exited | +|---|---|---| +| `kurrentdb_current_incoming_grpc_calls` | **the guard** | 0 / **2** / 0 | +| `kurrentdb_kestrel_connections` | reported only | 1 / 2 / 1 | + +`kurrentdb_current_incoming_grpc_calls` excludes the tool automatically: the guard's own request is plain +HTTP, not a gRPC call. `kurrentdb_kestrel_connections` counts that scrape — and would count a Prometheus +scraper too — so it is printed but never gated on, because a false refusal here has no override. + +If the metric is **absent**, the tool refuses. A renamed counter must fail loudly, not silently switch the +guard off. + +**Known limit:** it counts *open calls*, not *connected clients*. A client sitting idle with nothing in flight +reads as zero. In practice the real case is caught — an app holds a persistent `$all` subscription, and the +sibling-container guard covers the app being up at all. + +## The script + +The contract is the one [Kurrent Replicator](https://docs.kurrent.io/) documents, so a script written for +either tool runs on both. + +- The file may define `function transform(original)`. +- `original` has `Stream`, `EventType`, `Data` (the payload as an object), `Metadata` (an object or + `undefined`), plus read-only `EventId`, `EventNumber` and `Created` (an ISO-8601 string, so + `e.Created < "2026-08-25"` orders correctly). +- It returns an object of the same shape. Returning `undefined`, or an object with an empty `Stream` or + `EventType`, **drops** the event. +- `log.debug/info/warn/error(template, ...values)` writes to the tool's log. + +On top of the plain function, helpers keep the common repairs to one line each. Helpers and `transform` may be +combined: helpers run first, and `transform` sees the survivors. + +```js +dropStream("StartPipelineCommand-325888b9-621a-4f36-b091-40b8432799b2"); +dropStream(/^Offer-of-/); +dropStream(s => s.startsWith("Recording-") && s.endsWith("-test")); +dropEvent(e => e.EventId === "8d8066f9-bfbb-423e-a990-3fa5f7c6d4d6"); +dropEvent(e => e.EventType === "StopPipelineCommand" && e.Created < "2026-08-25"); +update("SetPipelineProperties", e => { e.Data.Properties["pyl1.hdr-profile"] = 0; return e; }); +updateById("c1ceefa2-363b-44ce-93a8-c54a0517550f", e => { e.Metadata.CorrelationId = e.Metadata.CausationId; return e; }); +renameType("PipelineStarted", "PipelineStartedV2"); +renameStream("Pipeline-old-id", "Pipeline-new-id"); +``` + +Order within one event: `dropEvent` predicates → `update` / `updateById` handlers → `transform`. +`dropStream`, `renameStream` and `renameType` are applied before any of that, so a dropped stream never +reaches a handler at all. + +**Sandbox.** The script runs in Jint with **no CLR access** — it cannot reach a .NET type, the filesystem or +the network. Recursion is capped at 64 frames, and all script code run for one event shares a single +**2-second** budget (not 2 seconds per call). + +**Only JSON payloads are transformable.** A payload that is not JSON — or that claims to be and is not — +bypasses the script entirely, is copied byte-for-byte, and is counted in the report. It is never dropped. + +**Numbers beyond 2^53.** JavaScript has one number type. A payload **no rule touched** is copied byte-for-byte +and is safe. The moment any rule modifies an event, its whole payload is re-rendered from JavaScript doubles, +and an integer larger than 2^53 (`9007199254740993` → `9007199254740992`) or a non-canonical literal (`1.0` → +`1`) changes. This is inherent to the script contract, not to this implementation. If your events carry int64 +identifiers, restrict your rules to the streams that need them. + +## Exit codes + +| Code | Meaning | State of your store | +|---|---|---| +| `0` | The rewrite completed and was verified. | The container runs on the new store; the old one is in `data.bak.`. | +| `1` | A guard refused, or you declined at the prompt — including a store on a named volume. | Untouched. Nothing was created or started. | +| `2` | The script does not parse (the message carries line and column), or an argument this version does not support (`--script` with `--eval`; `--force-volume-copy`). | Untouched. No container was started. | +| `3` | Docker unreachable, no such container, or the image could not be pulled. | Untouched. | +| `4` | The copy engine or the verification failed. | **Untouched.** The scratch store is removed. | +| `5` | The filesystem swap, or something after it, failed. | **Restored.** The original directory is back and the container is running on it. | + +## Rollback and status + +```bash +mp-rewrite my-eventstore --status # where the data is, which backups exist, whether it is safe now +mp-rewrite my-eventstore --rollback # put the most recent backup back +mp-rewrite my-eventstore --rollback /var/docker/data/data.bak.20260907T151200 +``` + +A rollback is itself reversible: the rewritten store is moved to `data.rolledback.`, not deleted. + +## Permissions, and why a helper container appears + +The KurrentDB image runs as **uid 1001**. The operator running this tool usually does not. + +- **The swap works anyway.** Renaming `data` needs write permission on its *parent* directory, not on `data` + itself, so directories full of files owned by another uid swap fine. +- **The new directory is made group/other writable** before the scratch store is pointed at it, and keeps + those permissions after the swap — that is what lets the original container go on writing. +- **Deleting `data.new.` needs root.** The store contains `index/stream-existence/` created with the + container's own umask and owner, which the tool's user cannot remove. So on a dry run, and on any failure + before the swap, the directory is removed by `rm -rf` inside a short-lived container from the same image, + running as root, with the parent bind-mounted. **The tool never needs root on the host.** That helper is + aimed only at a `.new.` directory this run created. + +## Limits + +- **Named volumes are refused.** The tool swaps stores by renaming directories, which needs a bind mount. + A store on a named volume is refused with exit `1` — a guard on the store's state, like the other refusals. + `--force-volume-copy` is **reserved** for a future version that implements the named-volume path; passing it + to this one exits `2` (an argument this version does not support, like `--script` with `--eval`) and says + so, rather than being silently ignored. +- **Integers beyond 2^53** change in any payload a rule touches (see *The script* above). +- **The connected-client guard counts open gRPC calls**, not connected clients (see *Refusals* above). + +## Not in scope + +Live replication while the application runs (other containers are stopped by contract), editing binary or +protobuf payloads, and clusters — one node in one container. diff --git a/src/MicroPlumberd.Rewrite/Report.cs b/src/MicroPlumberd.Rewrite/Report.cs new file mode 100644 index 0000000..36e15d4 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/Report.cs @@ -0,0 +1,104 @@ +using System.Text; +using MicroPlumberd.Migration; + +namespace MicroPlumberd.Rewrite; + +///

What the run did, in the form an operator reads before deciding whether to trust it. +public sealed record RewriteReport +{ + /// The exit code the process returns. + public required ExitCode Code { get; init; } + + /// One line saying what happened, shown first. + public required string Headline { get; init; } + + /// The container that was rewritten. + public StoreContainer? Container { get; init; } + + /// True when nothing was written. + public bool DryRun { get; init; } + + /// The migration id recorded in the new store's history. + public string? MigrationId { get; init; } + + /// sha256 of the script text, or null for a pure copy. + public string? ScriptChecksum { get; init; } + + /// The rules the run compiled to, in order. + public IReadOnlyList Descriptors { get; init; } = []; + + /// Source streams the rules dropped entirely, with their event counts. + public IReadOnlyList<(string Stream, long Events)> DroppedStreams { get; init; } = []; + + /// Source streams the rules touched or copied, with source and destination counts. + public IReadOnlyList<(string Stream, string Target, long Source, long Kept)> AffectedStreams { get; init; } = []; + + /// Events read, written, dropped and copied-verbatim-because-unreadable. + public long SourceEvents { get; init; } + + /// Events written to the new store. + public long Kept { get; init; } + + /// Events the rules dropped. + public long Dropped { get; init; } + + /// Events whose payload could not be parsed and were copied byte-for-byte. + public long UnparseableVerbatim { get; init; } + + /// User projections pre-created on the new store. + public IReadOnlyList CopiedProjections { get; init; } = []; + + /// Where the pre-rewrite store was kept. + public string? BackupDir { get; init; } + + /// The verification report, when one ran. + public VerificationReport? Verification { get; init; } + + /// How long the run took. + public TimeSpan Elapsed { get; init; } + + /// Extra lines (guard results, status listing). + public IReadOnlyList Notes { get; init; } = []; + + /// Renders the report. + public string Format() + { + var sb = new StringBuilder(); + sb.AppendLine(Headline); + if (Container is not null) + { + sb.AppendLine($" container : {Container.Name} ({Container.Image})"); + sb.AppendLine($" data : {Container.Data.StoreDir ?? "volume:" + Container.Data.VolumeName}"); + } + if (DryRun) sb.AppendLine(" mode : DRY RUN — nothing was written"); + if (MigrationId is not null) sb.AppendLine($" migration : {MigrationId}"); + if (ScriptChecksum is not null) sb.AppendLine($" script sha256: {ScriptChecksum}"); + foreach (var d in Descriptors) sb.AppendLine($" rule : {d}"); + + if (AffectedStreams.Count > 0) + { + sb.AppendLine($" affected streams ({AffectedStreams.Count}):"); + foreach (var (s, t, src, kept) in AffectedStreams.OrderBy(x => x.Stream, StringComparer.Ordinal)) + sb.AppendLine(s == t + ? $" {s}: {src} → {kept}" + : $" {s} → {t}: {src} → {kept}"); + } + if (DroppedStreams.Count > 0) + { + sb.AppendLine($" dropped streams ({DroppedStreams.Count}):"); + foreach (var (s, n) in DroppedStreams.OrderBy(x => x.Stream, StringComparer.Ordinal)) + sb.AppendLine($" {s} ({n} event(s))"); + } + + sb.AppendLine($" events : read {SourceEvents}, written {Kept}, dropped {Dropped}, " + + $"unreadable-copied-verbatim {UnparseableVerbatim}"); + if (CopiedProjections.Count > 0) + sb.AppendLine($" projections : {string.Join(", ", CopiedProjections)}"); + if (BackupDir is not null) sb.AppendLine($" backup : {BackupDir}"); + foreach (var n in Notes) sb.AppendLine($" {n}"); + if (Verification is not null) sb.Append(Verification.Format()); + sb.AppendLine($" elapsed : {Elapsed.TotalSeconds:0.0}s"); + sb.AppendLine($" exit code : {(int)Code} ({Code})"); + return sb.ToString(); + } +} diff --git a/src/MicroPlumberd.Rewrite/RewriteCommand.cs b/src/MicroPlumberd.Rewrite/RewriteCommand.cs new file mode 100644 index 0000000..672e439 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/RewriteCommand.cs @@ -0,0 +1,629 @@ +using System.Diagnostics; +using Docker.DotNet; +using KurrentDB.Client; +using MicroPlumberd.Migration; +using MicroPlumberd.Migration.Scripting; +using Microsoft.Extensions.Logging; + +namespace MicroPlumberd.Rewrite; + +/// +/// The tool itself. only parses arguments into and calls +/// this, so the end-to-end suite drives exactly the code a real invocation runs. +/// +public sealed class RewriteCommand +{ + /// How long the scratch store gets to come up. + private static readonly TimeSpan ScratchStartTimeout = TimeSpan.FromSeconds(120); + + /// How long the original container gets to come back after the swap. + private static readonly TimeSpan RestartTimeout = TimeSpan.FromSeconds(120); + + /// How long the tool waits for its own connections to drain before the pre-swap guard re-check. + private static readonly TimeSpan OwnCallDrainTimeout = TimeSpan.FromSeconds(15); + + /// The standard projection the copy engine needs live on the destination. + private const string ByEventType = "$by_event_type"; + + /// Streams the copy engine reserves for itself — excluded from the independent recount too. + private static readonly IReadOnlySet ReservedStreams = + new HashSet(StringComparer.Ordinal) { MigrationRunner.HistoryStreamName }; + + /// Runs the tool. Never throws for an expected failure — the outcome is in the report's code. + public static async Task RunAsync(RewriteOptions options, IDockerClient docker, + ILoggerFactory loggerFactory, CancellationToken ct = default) + { + ArgumentNullException.ThrowIfNull(options); + ArgumentNullException.ThrowIfNull(docker); + ArgumentNullException.ThrowIfNull(loggerFactory); + + var logger = loggerFactory.CreateLogger(); + var sw = Stopwatch.StartNew(); + + // Refused before ANY docker call, in every mode: the answer does not depend on inspecting anything, + // and an operator must not watch the tool start work it will not finish. + if (options.ForceVolumeCopy) + { + var message = "named-volume rewrite is not implemented in this version; move the store to a bind " + + "mount or wait for a version that implements --force-volume-copy"; + logger.LogError("{Message}", message); + // Exit 2 — an ARGUMENT this version does not support, like --script together with --eval. That is + // a different thing from the store itself being on a named volume, which is a guard on the store's + // state and exits 1 (RequireSwappableData below). Same subject, two refusals, two codes. + return new RewriteReport { Code = ExitCode.ScriptError, Headline = message, Elapsed = sw.Elapsed }; + } + + try + { + return options.Mode switch + { + RewriteMode.Status => await StatusAsync(options, docker, logger, sw, ct).ConfigureAwait(false), + RewriteMode.Rollback => await RollbackAsync(options, docker, loggerFactory, logger, sw, ct) + .ConfigureAwait(false), + _ => await RewriteAsync(options, docker, loggerFactory, logger, sw, ct).ConfigureAwait(false) + }; + } + catch (RewriteRefusedException ex) + { + logger.LogError("{Message}", ex.Message); + return new RewriteReport { Code = ex.Code, Headline = ex.Message, Elapsed = sw.Elapsed }; + } + } + + // ================================================================= rewrite + + private static async Task RewriteAsync(RewriteOptions options, IDockerClient docker, + ILoggerFactory lf, ILogger logger, Stopwatch sw, CancellationToken ct) + { + // STEP 0 — the script is parsed BEFORE anything else. A typo must cost the operator nothing: no + // container started, no store touched, no directory created. + // A run ALWAYS has a migration, even with no script. requirements.md § Safety and lead decision 4 both + // say every run records history, and a pure copy is the tool's headline invocation — it must not be the + // one that leaves no trace. A script-less run is modelled as a script that is empty rather than as "no + // migration": the id still carries the run's timestamp, the checksum is honestly sha256("") and stays + // comparable with a scripted run, and the recorded descriptor list is empty rather than absent. + ScriptMigration migration; + string? source; + try + { + source = options.ReadScriptSource(); + migration = new ScriptMigration(source ?? "", ScriptIdPrefix(DateTime.UtcNow), logger); + } + catch (Exception ex) when (ex is ScriptSyntaxException or FileNotFoundException or ArgumentException) + { + return new RewriteReport { Code = ExitCode.ScriptError, Headline = ex.Message, Elapsed = sw.Elapsed }; + } + + var store = new DockerStore(docker, logger); + + // STEP 1 — inspect. + var c = await store.InspectAsync(options.Container, ct).ConfigureAwait(false); + RequireSwappableData(c, options); + var (pathRefusals, pathNotes) = DockerStore.CheckStatePathsInsideMount(c.Env, c.Data); + if (pathRefusals.Count > 0) + return new RewriteReport + { + Code = ExitCode.GuardRefusal, Container = c, Elapsed = sw.Elapsed, Notes = pathNotes, + Headline = "Refusing: " + string.Join(" ", pathRefusals) + }; + var connectionString = DockerStore.ConnectionString(c, options.User, options.Password); + + // STEP 2 — guards, BEFORE the tool opens its own client to the old store, so the connection count it + // reads is the operator's, never its own. + var guards = await EvaluateGuardsAsync(store, c, connectionString, ct).ConfigureAwait(false); + if (guards.Refusal is not null) + return new RewriteReport + { + Code = ExitCode.GuardRefusal, Headline = guards.Refusal, Container = c, + Notes = guards.Notes, Elapsed = sw.Elapsed + }; + + // M1 — anchor the source NOW, while the guards have just said nobody is writing. Everything after + // this point (an unbounded confirmation prompt, then the whole copy) happens with the old store up and + // writable, so this position is what turns "nobody was writing then" into "nobody wrote at all". + ulong anchorBefore; + await using (var anchorClient = new KurrentDBClient(KurrentDBClientSettings.Create(connectionString))) + anchorBefore = await RewriteVerification.ReadSourceHeadAsync(anchorClient, ReservedStreams, ct) + .ConfigureAwait(false); + + // The plan, then the operator's confirmation. + var stamp = DirectorySwap.Stamp(DateTime.UtcNow); + var descriptorRecorder = new DescriptorRecorder(); + migration.Migrate(descriptorRecorder); + var descriptors = descriptorRecorder.Descriptors; + if (!options.Yes && !await ConfirmAsync(options, c, migration, descriptors, ct).ConfigureAwait(false)) + return new RewriteReport + { + Code = ExitCode.GuardRefusal, Headline = "Cancelled at the confirmation prompt; nothing changed.", + Container = c, Elapsed = sw.Elapsed + }; + + await store.EnsureImageAsync(c.Image, ct).ConfigureAwait(false); + + var swap = new DirectorySwap(c.Data, logger); + var newStoreDir = swap.CreateNewStoreDir(stamp); + var scratchName = $"mp-rewrite-{c.Name}-{stamp}"; + ScratchStore? scratch = null; + SwapResult? swapped = null; + KurrentDBClient? sourceClient = null; + KurrentDBProjectionManagementClient? sourceProjections = null; + + try + { + // STEP 3 — the scratch store. A dry run starts one too: the engine's destination pre-checks need + // it, and running the two modes down one code path is what makes a dry run evidence about the + // real one. + scratch = await store.StartScratchAsync(c, newStoreDir, scratchName, ct).ConfigureAwait(false); + await StoreHealth.WaitLiveAsync(scratch.ConnectionString, ScratchStartTimeout, logger, ct) + .ConfigureAwait(false); + + await using var destClient = new KurrentDBClient(KurrentDBClientSettings.Create(scratch.ConnectionString)); + await using var destProjections = + new KurrentDBProjectionManagementClient(KurrentDBClientSettings.Create(scratch.ConnectionString)); + await StoreHealth.WaitProjectionRunningAsync(destProjections, ByEventType, ScratchStartTimeout, logger, ct) + .ConfigureAwait(false); + + // NOT `await using`: these must be disposed BEFORE the pre-swap guard re-check below, or the tool + // counts its own gRPC calls as a connected client and refuses itself. The finally block disposes + // them on every other path. + sourceClient = new KurrentDBClient(KurrentDBClientSettings.Create(connectionString)); + sourceProjections = + new KurrentDBProjectionManagementClient(KurrentDBClientSettings.Create(connectionString)); + + var projectionCopy = options.NoProjectionCopy + ? null + : new ProjectionCopyContext + { + SourceProjections = sourceProjections, + DestProjections = destProjections, + SourceConnectionString = connectionString + }; + + // STEP 4/5 — copy and verify. Anything that fails here leaves the OLD store untouched. + MigrationRunResult run; + try + { + run = await new MigrationRunner(lf).RunAsync(sourceClient, destClient, + [migration], options.DryRun, projectionCopy, null, ct) + .ConfigureAwait(false); + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger.LogError(ex, "The copy failed; the old store was not touched."); + return new RewriteReport + { + Code = ExitCode.EngineFailure, Container = c, DryRun = options.DryRun, + Headline = $"Copy failed, the old store is untouched: {ex.Message}", Elapsed = sw.Elapsed + }; + } + + // STEP 5 — verify. Two independent things must both hold before an irreversible swap: + // (a) what the engine believed it wrote arrived intact (the library's per-stream counts + hash); + // (b) what the engine believed it READ is what the source actually holds, everything it read was + // either kept or dropped, and every stream that vanished is attributable to a rule. + // (a) alone cannot fail for a lost event: both its sides come from the same bookkeeping. + var recount = await RewriteVerification + .RecountSourceAsync(sourceClient, ReservedStreams, ct).ConfigureAwait(false); + var issues = RewriteVerification.Check(run.Copy, recount, descriptorRecorder.Reach); + + if (run.Verification is { AllOk: false } || issues.Count > 0) + { + foreach (var i in issues) logger.LogError("Verification: {Stream}: {Reason}", i.Subject, i.Reason); + return Fill(new RewriteReport + { + Code = ExitCode.EngineFailure, Container = c, + Headline = "Verification MISMATCH — the swap was not performed and the old store is untouched.", + Verification = run.Verification, Elapsed = sw.Elapsed, + Notes = issues.Select(i => $"unexplained : {i.Subject}: {i.Reason}").ToList() + }, run, migration, descriptors); + } + + if (options.DryRun) + return Fill(new RewriteReport + { + Code = ExitCode.Ok, Container = c, DryRun = true, + Headline = $"Dry run complete — {run.Copy.SourceEvents} event(s) read, " + + $"{run.Copy.Kept} would be written, {run.Copy.Dropped} dropped. Nothing changed.", + Elapsed = sw.Elapsed + }, run, migration, descriptors); + + // M1 — the guards ran minutes ago, before an unbounded prompt and the whole copy, with the old + // store writable throughout. Re-check both halves NOW, while nothing has been swapped and a + // refusal therefore costs nothing. + await sourceProjections.DisposeAsync().ConfigureAwait(false); + await sourceClient.DisposeAsync().ConfigureAwait(false); + + // Our own reads have to be gone before the connected-client guard runs again, or the tool refuses + // itself. Best-effort and bounded: if the count never settles, the guard below says so — which is + // the correct answer when someone else really is connected. + await WaitForOwnCallsToDrainAsync(connectionString, logger, ct).ConfigureAwait(false); + + var preSwapGuards = await EvaluateGuardsAsync(store, c, connectionString, ct).ConfigureAwait(false); + if (preSwapGuards.Refusal is not null) + return new RewriteReport + { + Code = ExitCode.GuardRefusal, Container = c, Elapsed = sw.Elapsed, + Notes = preSwapGuards.Notes, + Headline = "Refusing at the last check before the swap: " + preSwapGuards.Refusal + + " Nothing was swapped; the old store is untouched." + }; + + // Read LAST, after the drain wait and the re-guard, so the gap between the final look at the + // source and the stop is one round trip. Reading it before those (a wait of up to 15 s plus a + // guard evaluation) left a window in which a client could write, disconnect, and be missed by + // both checks. A fresh short-lived client: the guard above has already run, so this call cannot + // be counted against us. + ulong anchorAfter; + await using (var anchorClient = new KurrentDBClient(KurrentDBClientSettings.Create(connectionString))) + anchorAfter = await RewriteVerification.ReadSourceHeadAsync(anchorClient, ReservedStreams, ct) + .ConfigureAwait(false); + + if (anchorAfter != anchorBefore) + return Fill(new RewriteReport + { + Code = ExitCode.EngineFailure, Container = c, Elapsed = sw.Elapsed, + Headline = "The source was written to DURING the run (its last commit position moved from " + + $"{anchorBefore} to {anchorAfter}). Those events are not in the new store and " + + "the swap would destroy them — aborted, the old store is untouched.", + Verification = run.Verification + }, run, migration, descriptors); + + // STEP 6 — the swap. Both stores must be stopped first, and the scratch container must be gone + // before its data directory is renamed out from under it. + await store.StopAsync(scratch.Id, ct).ConfigureAwait(false); + await store.RemoveAsync(scratch.Id, ct).ConfigureAwait(false); + scratch = null; + + await store.StopAsync(c.Id, ct).ConfigureAwait(false); + + try + { + swapped = swap.Swap(newStoreDir, stamp); + } + catch (Exception ex) when (ex is IOException or UnauthorizedAccessException) + { + await store.StartAsync(c.Id, ct).ConfigureAwait(false); + return new RewriteReport + { + Code = ExitCode.FilesystemFailure, Container = c, + Headline = $"The swap failed and the original data is in place: {ex.Message}", + Elapsed = sw.Elapsed + }; + } + + if (options.FailAfterSwapForTest) + throw new InvalidOperationException( + "MP_REWRITE_TEST_FAIL_AFTER_SWAP: injected failure immediately after the swap."); + + await store.StartAsync(c.Id, ct).ConfigureAwait(false); + await StoreHealth.WaitLiveAsync(connectionString, RestartTimeout, logger, ct).ConfigureAwait(false); + + await using var afterProjections = + new KurrentDBProjectionManagementClient(KurrentDBClientSettings.Create(connectionString)); + var faulted = await StoreHealth.FaultedProjectionsAsync(afterProjections, ct).ConfigureAwait(false); + if (faulted.Count > 0) + throw new InvalidOperationException( + $"After the swap these projections are Faulted: {string.Join(", ", faulted)}."); + + return Fill(new RewriteReport + { + Code = ExitCode.Ok, Container = c, + Headline = $"Rewrite complete — {c.Name} is running on the new store.", + BackupDir = swapped.BackupDir, Verification = run.Verification, Elapsed = sw.Elapsed + }, run, migration, descriptors); + } + catch (Exception ex) when (ex is not OperationCanceledException and not RewriteRefusedException) + { + // STEP 6b — anything after the swap puts the original data back and starts the container on it. + if (swapped is not null) + { + await SafeStopAsync(store, c.Id, logger, ct).ConfigureAwait(false); + swap.Restore(swapped); + await PurgeNewStoreDirAsync(store, c.Image, swapped.NewStoreDir, logger, ct).ConfigureAwait(false); + await store.StartAsync(c.Id, ct).ConfigureAwait(false); + await SafeWaitLiveAsync(connectionString, logger, ct).ConfigureAwait(false); + return new RewriteReport + { + Code = ExitCode.FilesystemFailure, Container = c, + Headline = $"Failed after the swap; the ORIGINAL store has been restored and {c.Name} is " + + $"running on it: {ex.Message}", + Elapsed = sw.Elapsed + }; + } + logger.LogError(ex, "The rewrite failed before the swap."); + return new RewriteReport + { + Code = ExitCode.EngineFailure, Container = c, DryRun = options.DryRun, + Headline = $"Failed before the swap, the old store is untouched: {ex.Message}", Elapsed = sw.Elapsed + }; + } + finally + { + // The scratch container and its directory never outlive the run — a dry run included (E2E-06). + if (sourceProjections is not null) await SafeDisposeAsync(sourceProjections, logger).ConfigureAwait(false); + if (sourceClient is not null) await SafeDisposeAsync(sourceClient, logger).ConfigureAwait(false); + if (scratch is not null) await store.RemoveAsync(scratch.Id, ct).ConfigureAwait(false); + if (swapped is null) + await PurgeNewStoreDirAsync(store, c.Image, newStoreDir, logger, ct).ConfigureAwait(false); + } + } + + // ================================================================= status / rollback + + private static async Task StatusAsync(RewriteOptions options, IDockerClient docker, + ILogger logger, Stopwatch sw, CancellationToken ct) + { + var store = new DockerStore(docker, logger); + var c = await store.InspectAsync(options.Container, ct).ConfigureAwait(false); + + var notes = new List(); + string? safe; + if (c.Data.IsBind) + { + var swap = new DirectorySwap(c.Data, logger); + var backups = swap.Backups(); + notes.Add($"backups : {(backups.Count == 0 ? "(none)" : string.Join(", ", backups))}"); + + // Debris from an interrupted run. It is root-owned, an operator on this fleet cannot delete it by + // hand, and until now nothing told them it was there at all. + var leftovers = swap.NewStoreDirs(); + notes.Add($"interrupted : {(leftovers.Count == 0 ? "(none)" : string.Join(", ", leftovers) + + " — left by an interrupted run; remove with a root helper, they are not the operator's to delete")}"); + var scratch = await store.FindScratchContainersAsync(c.Name, ct).ConfigureAwait(false); + notes.Add($"scratch : {(scratch.Count == 0 ? "(none)" : string.Join(", ", scratch) + + " — a rewrite is running, or one was interrupted")}"); + + var (_, statusPathNotes) = DockerStore.CheckStatePathsInsideMount(c.Env, c.Data); + notes.AddRange(statusPathNotes); + + var guards = await EvaluateGuardsAsync(store, c, DockerStore.ConnectionString(c, options.User, options.Password), ct) + .ConfigureAwait(false); + notes.AddRange(guards.Notes); + safe = guards.Refusal is null ? "yes" : $"no — {guards.Refusal}"; + } + else + { + notes.Add($"backups : (not applicable — named volume '{c.Data.VolumeName}')"); + safe = "no — the store is on a named volume; --force-volume-copy is required"; + } + notes.Add($"safe : {safe}"); + + return new RewriteReport + { + Code = ExitCode.Ok, Container = c, Notes = notes, Elapsed = sw.Elapsed, + Headline = $"Status of {c.Name}:" + }; + } + + private static async Task RollbackAsync(RewriteOptions options, IDockerClient docker, + ILoggerFactory lf, ILogger logger, Stopwatch sw, CancellationToken ct) + { + var store = new DockerStore(docker, logger); + var c = await store.InspectAsync(options.Container, ct).ConfigureAwait(false); + RequireSwappableData(c, options); + + // A rollback discards every write made since the backup, so it is MORE destructive than a rewrite — + // and it was the one path with no sibling and no connected-client check at all. + var guards = await EvaluateGuardsAsync(store, c, + DockerStore.ConnectionString(c, options.User, options.Password), ct).ConfigureAwait(false); + if (guards.Refusal is not null) + return new RewriteReport + { + Code = ExitCode.GuardRefusal, Container = c, Elapsed = sw.Elapsed, Notes = guards.Notes, + Headline = guards.Refusal + }; + + var swap = new DirectorySwap(c.Data, logger); + var stamp = DirectorySwap.Stamp(DateTime.UtcNow); + + // Only a directory this tool created as a backup may be restored. An arbitrary path would let a typo + // rename something else into the store's place — and the store directory is not a thing to guess at. + if (options.BackupDir is { } requested) + { + var known = swap.Backups(); + var match = known.FirstOrDefault(b => + string.Equals(Path.GetFullPath(b).TrimEnd(Path.DirectorySeparatorChar), + Path.GetFullPath(requested).TrimEnd(Path.DirectorySeparatorChar), StringComparison.Ordinal)); + if (match is null) + return new RewriteReport + { + Code = ExitCode.GuardRefusal, Container = c, Elapsed = sw.Elapsed, + Headline = $"'{requested}' is not one of this store's backups. Available: " + + (known.Count == 0 ? "(none)" : string.Join(", ", known)) + }; + } + + var wasRunning = await store.IsRunningAsync(c.Id, ct).ConfigureAwait(false); + if (wasRunning) await store.StopAsync(c.Id, ct).ConfigureAwait(false); + + RollbackResult result; + try + { + result = swap.Rollback(options.BackupDir, stamp); + } + catch (Exception ex) when (ex is IOException or UnauthorizedAccessException) + { + await store.StartAsync(c.Id, ct).ConfigureAwait(false); + return new RewriteReport + { + Code = ExitCode.FilesystemFailure, Container = c, Elapsed = sw.Elapsed, + Headline = $"Rollback failed and the store was left as it was: {ex.Message}" + }; + } + + await store.StartAsync(c.Id, ct).ConfigureAwait(false); + await StoreHealth.WaitLiveAsync(DockerStore.ConnectionString(c, options.User, options.Password), + RestartTimeout, logger, ct).ConfigureAwait(false); + + return new RewriteReport + { + Code = ExitCode.Ok, Container = c, Elapsed = sw.Elapsed, BackupDir = result.RestoredFrom, + Headline = $"Rolled back — {c.Name} is running on {result.RestoredFrom}.", + Notes = [$"rewritten store kept at: {result.DisplacedTo}"] + }; + } + + // ================================================================= helpers + + /// The id prefix the tool gives every run, so a second pure copy is its own history record. + /// + /// Millisecond precision, not second: two runs of the same script within one second would otherwise share + /// an id, and the history guard would silently SKIP the second as already applied — the trap is removed + /// rather than documented. + /// + public static string ScriptIdPrefix(DateTime utc) => $"rewrite_{utc:yyyyMMddTHHmmssfff}"; + + /// + /// Refuses a store this tool cannot swap by renaming directories. + /// + /// + /// Exit 1, a guard refusal on the STORE'S STATE — distinct from passing --force-volume-copy, which + /// is an unsupported argument and exits 2. Internal so both codes can be pinned by a test. + /// + internal static void RequireSwappableData(StoreContainer c, RewriteOptions options) + { + if (c.Data.IsBind) return; + throw new RewriteRefusedException(ExitCode.GuardRefusal, + $"The store of '{c.Name}' is on the named volume '{c.Data.VolumeName}', which this tool cannot " + + "swap by renaming directories. Move the store onto a bind mount."); + } + + private sealed record GuardOutcome(string? Refusal, IReadOnlyList Notes); + + /// + /// The two refusals with no override (ADR 7): a sibling container of the same compose project is running, + /// or a client has work in flight against the store. + /// + private static async Task EvaluateGuardsAsync(DockerStore store, StoreContainer c, + string connectionString, CancellationToken ct) + { + var notes = new List(); + + var siblings = await store.FindRunningSiblingsAsync(c, ct).ConfigureAwait(false); + notes.Add($"siblings : {(siblings.Count == 0 ? "none running" : string.Join(", ", siblings))}"); + + var (grpc, kestrel) = await DockerStore.ReadConnectionMetricsAsync(connectionString, ct) + .ConfigureAwait(false); + notes.Add($"clients : {DockerStore.OpenGrpcCallsMetric}={grpc} " + + $"({DockerStore.KestrelConnectionsMetric}={kestrel}, includes this tool's own scrape)"); + + if (siblings.Count > 0) + return new GuardOutcome( + $"Refusing: {siblings.Count} container(s) of compose project '{c.ComposeProject}' are still " + + $"running ({string.Join(", ", siblings)}). Stop them first — a writer mid-rewrite loses data.", + notes); + + if (grpc < 0) + return new GuardOutcome( + $"Refusing: the store does not expose '{DockerStore.OpenGrpcCallsMetric}', so the " + + "connected-client guard cannot be evaluated. Refusing beats rewriting under a live writer.", + notes); + + if (grpc > 0) + return new GuardOutcome( + $"Refusing: {grpc} gRPC call(s) are open against the store ({DockerStore.OpenGrpcCallsMetric}" + + $"={grpc}). Disconnect every client first — a writer mid-rewrite loses data.", + notes); + + return new GuardOutcome(null, notes); + } + + /// How the plan and the report name the script — a hash of nothing is not a useful thing to read. + private static string DescribeScript(ScriptMigration m) => + m.IsPureCopy ? $"(none — pure copy; recorded as {m.ScriptChecksum[..12]}…)" : m.ScriptChecksum; + + private static async Task ConfirmAsync(RewriteOptions options, StoreContainer c, + ScriptMigration migration, IReadOnlyList descriptors, CancellationToken ct) + { + var output = options.Output ?? Console.Out; + await output.WriteLineAsync($"About to rewrite the store of container '{c.Name}' ({c.Image}).") + .ConfigureAwait(false); + await output.WriteLineAsync($" data : {c.Data.StoreDir}").ConfigureAwait(false); + await output.WriteLineAsync($" script sha256 : {DescribeScript(migration)}").ConfigureAwait(false); + foreach (var d in descriptors) await output.WriteLineAsync($" rule : {d}").ConfigureAwait(false); + await output.WriteLineAsync("Type 'yes' to continue: ").ConfigureAwait(false); + + var input = options.ConfirmationInput ?? Console.In; + var answer = await input.ReadLineAsync(ct).ConfigureAwait(false); + return string.Equals(answer?.Trim(), "yes", StringComparison.OrdinalIgnoreCase); + } + + private static RewriteReport Fill(RewriteReport r, MigrationRunResult run, ScriptMigration migration, + IReadOnlyList descriptors) => r with + { + MigrationId = migration.Id, + ScriptChecksum = DescribeScript(migration), + Descriptors = descriptors, + SourceEvents = run.Copy.SourceEvents, + Kept = run.Copy.Kept, + Dropped = run.Copy.Dropped, + UnparseableVerbatim = run.Copy.UnparseableVerbatim, + CopiedProjections = run.CopiedProjections, + DroppedStreams = run.Copy.SourceStreams + .Where(kv => kv.Value.Kept == 0 && kv.Value.Dropped > 0) + .Select(kv => (kv.Key, kv.Value.Dropped)).ToList(), + AffectedStreams = run.Copy.SourceStreams + .Select(kv => (kv.Key, kv.Value.TargetStream, kv.Value.SourceCount, kv.Value.Kept)).ToList() + }; + + /// + /// Removes a .new. directory. The suffix check is the safety interlock on a root-container + /// rm -rf: this tool must only ever delete a directory IT created for this run. + /// + private static async Task PurgeNewStoreDirAsync(DockerStore store, string image, string dir, ILogger logger, + CancellationToken ct) + { + if (!Path.GetFileName(dir.TrimEnd(Path.DirectorySeparatorChar)).Contains(DirectorySwap.NewSuffix, + StringComparison.Ordinal)) + throw new InvalidOperationException( + $"Refusing to purge '{dir}': only a '{DirectorySwap.NewSuffix}' directory this run created " + + "may be removed."); + try + { + await store.PurgeDirectoryAsync(image, dir, ct).ConfigureAwait(false); + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + // Leftover scratch data is untidy, never dangerous — it must not mask the real outcome of the run. + logger.LogWarning(ex, "Could not remove the scratch directory {Dir}; remove it by hand.", dir); + } + } + + /// + /// Waits, briefly, for the tool's OWN gRPC calls to finish after it disposes its clients, so the + /// connected-client guard does not count them. Bounded and best-effort: if the count never reaches zero + /// the guard itself decides, which is the right answer when someone else is genuinely connected. + /// + private static async Task WaitForOwnCallsToDrainAsync(string connectionString, ILogger logger, + CancellationToken ct) + { + var deadline = DateTime.UtcNow + OwnCallDrainTimeout; + while (DateTime.UtcNow < deadline) + { + var (grpc, _) = await DockerStore.ReadConnectionMetricsAsync(connectionString, ct) + .ConfigureAwait(false); + if (grpc <= 0) return; + await Task.Delay(200, ct).ConfigureAwait(false); + } + logger.LogInformation("The store still reports open gRPC calls after {Seconds}s; the guard below " + + "decides whether they are ours or a client's.", OwnCallDrainTimeout.TotalSeconds); + } + + private static async Task SafeDisposeAsync(IAsyncDisposable d, ILogger logger) + { + try { await d.DisposeAsync().ConfigureAwait(false); } + catch (Exception ex) { logger.LogWarning(ex, "Disposing a store client failed."); } + } + + private static async Task SafeStopAsync(DockerStore store, string id, ILogger logger, CancellationToken ct) + { + try { if (await store.IsRunningAsync(id, ct).ConfigureAwait(false)) await store.StopAsync(id, ct).ConfigureAwait(false); } + catch (Exception ex) { logger.LogWarning(ex, "Could not stop {Id} while restoring.", id); } + } + + private static async Task SafeWaitLiveAsync(string connectionString, ILogger logger, CancellationToken ct) + { + try { await StoreHealth.WaitLiveAsync(connectionString, RestartTimeout, logger, ct).ConfigureAwait(false); } + catch (Exception ex) { logger.LogWarning(ex, "The restored store did not report live in time."); } + } +} diff --git a/src/MicroPlumberd.Rewrite/RewriteOptions.cs b/src/MicroPlumberd.Rewrite/RewriteOptions.cs new file mode 100644 index 0000000..1001c43 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/RewriteOptions.cs @@ -0,0 +1,90 @@ +namespace MicroPlumberd.Rewrite; + +/// What the operator asked the tool to do. +public enum RewriteMode +{ + /// Rewrite the store (the default). + Rewrite, + + /// Swap a backup back in and restart the container. + Rollback, + + /// Print the store's location, backups and guard results; change nothing. + Status +} + +/// +/// The parsed command line. builds one of these and does nothing else, so every test +/// drives in-process with exactly the inputs a real invocation produces. +/// +public sealed record RewriteOptions +{ + /// Docker container id or name of the KurrentDB to rewrite. + public required string Container { get; init; } + + /// Path to a .js rule script, or null. + public string? ScriptPath { get; init; } + + /// Inline script source (--eval), or null. + public string? Eval { get; init; } + + /// Copy nothing; report what a real run would do. + public bool DryRun { get; init; } + + /// Do not pre-create the app's user projections on the new store. + public bool NoProjectionCopy { get; init; } + + /// Skip the confirmation prompt. + public bool Yes { get; init; } + + /// + /// RESERVED. design.md §1 defines a named-volume swap through a helper container; this version does not + /// implement it, so setting this is REFUSED rather than ignored. + /// + /// + /// A flag that is accepted and does nothing is a trap: an operator on a named-volume host passes it, is + /// refused anyway, and has no way to tell that the flag was never going to help. The option is kept on the + /// command line only so it fails with the reason instead of "unknown argument". + /// + public bool ForceVolumeCopy { get; init; } + + /// Rewrite, rollback or status. + public RewriteMode Mode { get; init; } = RewriteMode.Rewrite; + + /// For : the backup directory to restore; null = the most recent. + public string? BackupDir { get; init; } + + /// + /// TEST HOOK. Throws immediately after the directory swap so the restore path can be exercised. Set from + /// MP_REWRITE_TEST_FAIL_AFTER_SWAP=1 by , or directly by a test. + /// + /// + /// It exists because the restore path is the one branch that only runs when something has already gone + /// wrong — the least-travelled code in the tool, and the one an operator most needs to work. + /// + public bool FailAfterSwapForTest { get; init; } + + /// Username for the store being rewritten. Defaults to the fleet's. + public string User { get; init; } = DockerStore.DefaultUser; + + /// Password for the store being rewritten. Defaults to the fleet's. + public string Password { get; init; } = DockerStore.DefaultPassword; + + /// Where the confirmation prompt reads its answer from. Defaults to standard input. + public TextReader? ConfirmationInput { get; init; } + + /// Where the tool writes its plan and report. Defaults to standard output. + public TextWriter? Output { get; init; } + + /// The script source, from or ; null for a pure copy. + public string? ReadScriptSource() + { + if (Eval is not null && ScriptPath is not null) + throw new ArgumentException("--script and --eval are mutually exclusive."); + if (Eval is not null) return Eval; + if (ScriptPath is null) return null; + if (!File.Exists(ScriptPath)) + throw new FileNotFoundException($"Script file not found: {ScriptPath}", ScriptPath); + return File.ReadAllText(ScriptPath); + } +} diff --git a/src/MicroPlumberd.Rewrite/RewriteVerification.cs b/src/MicroPlumberd.Rewrite/RewriteVerification.cs new file mode 100644 index 0000000..914e1b4 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/RewriteVerification.cs @@ -0,0 +1,166 @@ +using KurrentDB.Client; +using MicroPlumberd.Migration; + +namespace MicroPlumberd.Rewrite; + +/// What a compiled rule set can explain about a whole stream disappearing. +/// Predicates that drop a stream outright, asked by name. +/// Declared stream renames, so a drop rule can be asked about the renamed name too. +/// +/// True when a DropEvent or a generic Transform is registered. Either can empty any stream one +/// event at a time, and neither can be asked about a stream without replaying the copy — so their presence +/// makes every disappearance attributable, and the gate below is only decisive without them. +/// +public sealed record RuleReach( + IReadOnlyList> StreamDrops, + IReadOnlyList<(string From, string To)> StreamRenames, + bool HasEventLevelDrop) +{ + /// A rule set that explains nothing — what a pure copy has. + public static RuleReach None { get; } = new([], [], false); + + /// Whether some declared rule accounts for vanishing entirely. + public bool ExplainsLossOf(string stream) + { + if (HasEventLevelDrop) return true; + if (StreamDrops.Any(p => p(stream))) return true; + // A DropStream declared AFTER a RenameStream sees the new name, so ask about that too. + foreach (var (from, to) in StreamRenames) + if (string.Equals(from, stream, StringComparison.Ordinal) && StreamDrops.Any(p => p(to))) + return true; + return false; + } +} + +/// One thing the gate could not account for. Any of these aborts the run before the swap. +/// The stream it is about. +/// What does not add up, in the operator's terms. +public sealed record VerificationIssue(string Subject, string Reason); + +/// +/// The check `requirements.md` actually asks for: **did anything vanish that no rule dropped?** +/// +/// +/// The library's is worth having and is kept — it re-reads the +/// destination and catches truncation, reordering and corruption through a per-stream write hash. What it +/// cannot do is answer this question, because both sides of its comparison come from the copy engine's own +/// bookkeeping: an event the engine never counted lowers its expectation and its result together, and the +/// destination then matches exactly. A stream emptied by a bug is even reported as an *intended* drop. +/// So this gate brings its own facts. reads the SOURCE again and counts +/// per stream; compares that against what the engine believed it read, insists every event +/// was either kept or dropped, and requires a rule to account for any stream that vanished. +/// +public static class RewriteVerification +{ + /// + /// The $all commit position of the last event the copy would COPY — the source's head, ignoring the + /// system and link traffic that keeps moving on an idle store. + /// + /// + /// This is the run's anchor. The guards can only say "nobody was writing" at the instant they ran, and + /// between that instant and the swap sit an unbounded confirmation prompt and the whole copy, with the old + /// store up and writable. Re-reading this immediately before the swap is what turns "nobody was writing + /// then" into "nobody wrote at all" — and an event appended after the copy's read passed its position would + /// otherwise be destroyed by the swap, silently, because the engine never saw it and so its own + /// bookkeeping balances perfectly without it. + /// System streams and link events are skipped for a reason: an idle KurrentDB writes $stats + /// and its projections emit links continuously, so an anchor over raw $all would advance on its own + /// and refuse every run. + /// + public static async Task ReadSourceHeadAsync(KurrentDBClient source, + IReadOnlySet reservedStreams, CancellationToken ct = default) + { + await foreach (var re in source + .ReadAllAsync(Direction.Backwards, Position.End, resolveLinkTos: false, + cancellationToken: ct).ConfigureAwait(false)) + { + var er = re.Event; + if (er is null) continue; + if (er.EventStreamId.Length > 0 && er.EventStreamId[0] == '$') continue; + if (reservedStreams.Contains(er.EventStreamId)) continue; + if (er.EventType.Length > 0 && er.EventType[0] == '$') continue; + return er.Position.CommitPosition; + } + return 0; + } + + /// + /// Counts the source's copyable events per stream, independently of the copy engine. + /// + /// + /// The classification below deliberately RE-STATES the engine's rather than sharing it. Sharing would make + /// the recount inherit the very bug it exists to catch — a classifier that wrongly discards a stream would + /// discard it identically on both sides and the check would agree with itself. The cost is one extra + /// $all pass over an offline store, which is a cheap price for the last check before an + /// irreversible swap. + /// + public static async Task> RecountSourceAsync(KurrentDBClient source, + IReadOnlySet reservedStreams, CancellationToken ct = default) + { + var counts = new Dictionary(StringComparer.Ordinal); + var read = source.ReadAllAsync(Direction.Forwards, Position.Start, resolveLinkTos: false, + cancellationToken: ct); + await foreach (var re in read.ConfigureAwait(false)) + { + var er = re.Event; + if (er is null) continue; + if (er.EventStreamId.Length > 0 && er.EventStreamId[0] == '$') continue; + if (reservedStreams.Contains(er.EventStreamId)) continue; + if (er.EventType.Length > 0 && er.EventType[0] == '$') continue; // includes the "$>" link type + counts.TryGetValue(er.EventStreamId, out var c); + counts[er.EventStreamId] = c + 1; + } + return counts; + } + + /// + /// Compares the independent recount against the copy's own account of itself. An empty result means the + /// swap may proceed; anything in it must abort the run with the old store untouched. + /// + public static IReadOnlyList Check(CopyResult copy, + IReadOnlyDictionary trueSourceCounts, RuleReach rules) + { + ArgumentNullException.ThrowIfNull(copy); + ArgumentNullException.ThrowIfNull(trueSourceCounts); + ArgumentNullException.ThrowIfNull(rules); + + var issues = new List(); + var streams = trueSourceCounts.Keys + .Union(copy.SourceStreams.Keys, StringComparer.Ordinal) + .OrderBy(s => s, StringComparer.Ordinal); + + foreach (var stream in streams) + { + trueSourceCounts.TryGetValue(stream, out var actuallyThere); + copy.SourceStreams.TryGetValue(stream, out var info); + var engineSaw = info?.SourceCount ?? 0; + + // 1. The engine has to have SEEN everything that is in the source. This is the check the library's + // verifier structurally cannot make: a read that stops early, or a stream never enumerated, + // lowers its expectation by exactly as much as its result. + if (engineSaw != actuallyThere) + { + issues.Add(new VerificationIssue(stream, + $"the source holds {actuallyThere} copyable event(s) but the copy engine read {engineSaw}")); + continue; // the counts below are derived from a number already known to be wrong + } + + if (info is null || actuallyThere == 0) continue; + + // NOTE: there is deliberately no "SourceCount == Kept + Dropped" check here. It reads like a + // guard and is a TAUTOLOGY: CopyEngine increments SourceCount and then unconditionally exactly one + // of Dropped or Kept, so it can never fail. A check that cannot fail is worse than no check — + // it reports confidence it has not earned. + + // 2. A stream that vanished entirely must be attributable to a rule. Without this, a stream emptied + // by a bug is reported to the operator as an intended drop — and on a pure copy, where no rule + // can explain anything, that is exactly the silent total loss this tool must never cause. + if (info.Kept == 0 && !rules.ExplainsLossOf(stream)) + issues.Add(new VerificationIssue(stream, + $"the whole stream ({info.SourceCount} event(s)) is absent from the new store and no rule " + + "accounts for it")); + } + + return issues; + } +} diff --git a/src/MicroPlumberd.Rewrite/StoreHealth.cs b/src/MicroPlumberd.Rewrite/StoreHealth.cs new file mode 100644 index 0000000..5969160 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/StoreHealth.cs @@ -0,0 +1,139 @@ +using KurrentDB.Client; +using Microsoft.Extensions.Logging; + +namespace MicroPlumberd.Rewrite; + +/// Waiting for a KurrentDB node to be usable, and reporting when it never became so. +public static class StoreHealth +{ + /// A stream that cannot exist — reading it is the cheapest real gRPC round trip available. + private const string ProbeStream = "$$mp-rewrite-readiness-probe"; + + /// + /// Waits until the store is ready for the kind of call this tool actually makes: first /health/live, + /// then a real gRPC round trip. + /// + /// + /// `/health/live` alone is the wrong signal. It is served over HTTP/1.1 and becomes available + /// BEFORE KurrentDB will serve gRPC on the same port — measured at 528 ms of daylight on a freshly + /// started container (saturn, 2026-09-07). Every client this tool opens is gRPC, so a caller that trusted + /// the health check alone could open one inside that window and be answered with an HTTP/2 GOAWAY carrying + /// HTTP_1_1_REQUIRED, surfacing as RpcException(Internal). The client's own retry usually + /// hides it; on a loaded host it does not, and it took out a scenario's fixture on the 81-test run. + /// This is the single place every client in the tool and the suite waits, so the probe belongs here + /// rather than in each caller. + /// + public static async Task WaitLiveAsync(string connectionString, TimeSpan timeout, ILogger logger, + CancellationToken ct = default) + { + var url = new Uri(StoreHttp.BaseUriOf(connectionString), "health/live"); + var deadline = DateTime.UtcNow + timeout; + var last = "no response"; + var httpLive = false; + + while (!httpLive && DateTime.UtcNow < deadline) + { + ct.ThrowIfCancellationRequested(); + try + { + using var resp = await StoreHttp.GetAsync(connectionString, "health/live", ct) + .ConfigureAwait(false); + if (resp.IsSuccessStatusCode) + { + logger.LogInformation("{Url} is live.", url); + httpLive = true; + break; + } + last = $"HTTP {(int)resp.StatusCode}"; + } + catch (Exception ex) when (ex is HttpRequestException or TaskCanceledException) + { + last = ex.Message; + } + await Task.Delay(250, ct).ConfigureAwait(false); + } + + if (!httpLive) + throw new TimeoutException( + $"{url} did not become live within {timeout.TotalSeconds:0}s (last: {last})."); + + await WaitGrpcServableAsync(connectionString, deadline, timeout, logger, ct).ConfigureAwait(false); + } + + /// Polls a trivial read until the server actually serves gRPC, or the shared deadline passes. + private static async Task WaitGrpcServableAsync(string connectionString, DateTime deadline, TimeSpan timeout, + ILogger logger, CancellationToken ct) + { + var last = "no attempt"; + while (DateTime.UtcNow < deadline) + { + ct.ThrowIfCancellationRequested(); + try + { + // A fresh client per attempt: the client caches channel discovery, and a channel poisoned + // during the window would otherwise be reused for the rest of the run. + await using var probe = new KurrentDBClient(KurrentDBClientSettings.Create(connectionString)); + var read = probe.ReadStreamAsync(Direction.Backwards, ProbeStream, StreamPosition.End, + maxCount: 1, resolveLinkTos: false, cancellationToken: ct); + _ = await read.ReadState.ConfigureAwait(false); // StreamNotFound — the answer does not matter + logger.LogInformation("{Store} is serving gRPC.", StoreHttp.BaseUriOf(connectionString)); + return; + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + last = ex.Message.Split('\n')[0]; + } + await Task.Delay(200, ct).ConfigureAwait(false); + } + + throw new TimeoutException( + $"{StoreHttp.BaseUriOf(connectionString)} answers /health/live but did not serve a gRPC call " + + $"within {timeout.TotalSeconds:0}s (last: {last}). The health endpoint is HTTP/1.1 and becomes " + + "available before the gRPC endpoint does."); + } + + /// Polls until the named projection reports Running. + /// + /// The copy engine needs $by_event_type live on the destination, and mp-migrate has always + /// required it: without it the pre-created join projections have nothing to read and the merge streams + /// come out empty — a silent, plausible-looking result rather than a failure. + /// + public static async Task WaitProjectionRunningAsync(KurrentDBProjectionManagementClient projections, + string name, TimeSpan timeout, ILogger logger, CancellationToken ct = default) + { + var deadline = DateTime.UtcNow + timeout; + string? last = null; + while (DateTime.UtcNow < deadline) + { + ct.ThrowIfCancellationRequested(); + try + { + var s = await projections.GetStatusAsync(name, cancellationToken: ct).ConfigureAwait(false); + last = s?.Status; + if (last is not null && last.Contains("Running", StringComparison.OrdinalIgnoreCase)) + { + logger.LogInformation("Projection {Name} is {Status}.", name, last); + return; + } + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + last = ex.Message; + } + await Task.Delay(250, ct).ConfigureAwait(false); + } + throw new TimeoutException( + $"Projection '{name}' did not reach Running within {timeout.TotalSeconds:0}s (last: {last ?? "unknown"})."); + } + + /// Names of the projections currently reporting Faulted. + public static async Task> FaultedProjectionsAsync( + KurrentDBProjectionManagementClient projections, CancellationToken ct = default) + { + var faulted = new List(); + await foreach (var p in projections.ListAllAsync(cancellationToken: ct).ConfigureAwait(false)) + if (p.Status?.Contains("Faulted", StringComparison.OrdinalIgnoreCase) == true) + faulted.Add(p.Name); + return faulted; + } +} diff --git a/src/MicroPlumberd.Rewrite/StoreHttp.cs b/src/MicroPlumberd.Rewrite/StoreHttp.cs new file mode 100644 index 0000000..1c626f6 --- /dev/null +++ b/src/MicroPlumberd.Rewrite/StoreHttp.cs @@ -0,0 +1,53 @@ +using System.Net.Http.Headers; +using System.Text; +using MicroPlumberd; + +namespace MicroPlumberd.Rewrite; + +/// +/// The tool's HTTP access to a store's management endpoints, over ONE shared client. +/// +/// +/// Why this exists. KurrentHttpEndpoint.CreateClient builds a fresh +/// — and therefore a fresh connection pool — per call. That is fine for the +/// handful of calls the projection copier makes, but the connected-client guard is polled: once per guard +/// evaluation, and every 200 ms while the tool waits for its own connections to drain before the swap. Each +/// one leaked a handler and its sockets. +/// Measured 2026-09-07: the end-to-end suite aborted mid-run after 71 of 76 tests — and +/// dotnet test still printed Passed! for the 71 that had run, which is exactly the +/// "a dead run looks green" failure the CI test-count floor guards against. +/// One client, auth per request. is thread-safe and is meant to be shared. +/// +internal static class StoreHttp +{ + private static readonly HttpClient Client = new(new HttpClientHandler + { + // The tool talks to a store it has just been pointed at, over loopback or a docker bridge, usually + // with a self-signed certificate or none at all. + ServerCertificateCustomValidationCallback = HttpClientHandler.DangerousAcceptAnyServerCertificateValidator + }) + { + Timeout = TimeSpan.FromSeconds(30) + }; + + /// GETs a path relative to the store's HTTP base, with basic auth. + public static Task GetAsync(Uri baseUri, string user, string pass, string path, + CancellationToken ct = default) + { + var request = new HttpRequestMessage(HttpMethod.Get, new Uri(baseUri, path)); + request.Headers.Authorization = new AuthenticationHeaderValue("Basic", + Convert.ToBase64String(Encoding.ASCII.GetBytes($"{user}:{pass}"))); + return Client.SendAsync(request, ct); + } + + /// GETs a path using the credentials embedded in a KurrentDB connection string. + public static Task GetAsync(string connectionString, string path, + CancellationToken ct = default) + { + var (baseUri, user, pass) = KurrentHttpEndpoint.Parse(connectionString); + return GetAsync(baseUri, user, pass, path, ct); + } + + /// The store's HTTP base address, for messages. + public static Uri BaseUriOf(string connectionString) => KurrentHttpEndpoint.Parse(connectionString).BaseUri; +} diff --git a/src/MicroPlumberd.Services/ContainerExtensions.cs b/src/MicroPlumberd.Services/ContainerExtensions.cs index 45e31c8..6ebc0a3 100644 --- a/src/MicroPlumberd.Services/ContainerExtensions.cs +++ b/src/MicroPlumberd.Services/ContainerExtensions.cs @@ -173,11 +173,31 @@ public static IServiceCollection AddBackgroundServiceIfMissing(this IS /// The service collection. /// If true, uses persistent subscriptions; otherwise, uses catch-up subscriptions. /// The stream position to start reading from. If null, defaults to the start of the stream. + /// Where to source the merged stream. Default (unchanged). + /// is opt-in and CATCH-UP-only, with WEAKER CaughtUp semantics — see that + /// enum value: CaughtUp may fire before all history is delivered, so do NOT opt in if the read model uses + /// ICaughtUpHandler as a "fully caught up / now authoritative" readiness signal. No-loss + commit order still hold. /// The service collection for method chaining. public static IServiceCollection AddScopedEventHandler(this IServiceCollection services, - bool persistently = false, FromStream? start = null) where TEventHandler : class, IEventHandler, ITypeRegister + bool persistently = false, FromStream? start = null, MergeSource mergeSource = MergeSource.Projection) + where TEventHandler : class, IEventHandler, ITypeRegister { - return services.AddScoped().AddEventHandler(persistently, start); + return services.AddScoped().AddEventHandler(persistently, start, mergeSource); + } + + // Registration-time guards (fail fast): index-backing is CATCH-UP-only (persistent subscriptions cannot see + // index links — SPIKE-7, permanent), and only Start/End are meaningful on a filtered-$all subscription. + private static void ValidateMergeSource(MergeSource mergeSource, bool persistently, FromStream? start) + { + if (mergeSource != MergeSource.UserDefinedIndex) return; + if (persistently) + throw new InvalidOperationException( + "index-backed merge supports catch-up subscriptions only; persistent subscriptions stay " + + "projection-backed (SPIKE-7). Remove persistently:true or use MergeSource.Projection."); + if (start is { } s && s != FromStream.Start && s != FromStream.End) + throw new InvalidOperationException( + "index-backed merge supports only Start or End as a start position; a specific stream revision has " + + "no meaning on a filtered-$all subscription."); } /// @@ -189,13 +209,19 @@ public static IServiceCollection AddScopedEventHandler(this IServ /// The service collection. /// If true, uses persistent subscriptions; otherwise, uses catch-up subscriptions. /// The stream position to start reading from. If null, defaults to the start of the stream. + /// Where to source the merged stream. Default (unchanged). + /// is opt-in and CATCH-UP-only, with WEAKER CaughtUp semantics — see that + /// enum value: CaughtUp may fire before all history is delivered, so do NOT opt in if the read model uses + /// ICaughtUpHandler as a "fully caught up / now authoritative" readiness signal. No-loss + commit order still hold. /// The service collection for method chaining. public static IServiceCollection AddSingletonEventHandler(this IServiceCollection services, - bool persistently = false, FromStream? start = null) where TEventHandler : class, IEventHandler, ITypeRegister + bool persistently = false, FromStream? start = null, MergeSource mergeSource = MergeSource.Projection) + where TEventHandler : class, IEventHandler, ITypeRegister { + ValidateMergeSource(mergeSource, persistently, start); services.AddSingleton(); services.AddSingleton>(); - services.AddSingleton(sp => sp.GetRequiredService>().Configure(persistently, start)); + services.AddSingleton(sp => sp.GetRequiredService>().Configure(persistently, start, mergeSource)); // EventHandlerExecutor delegates directly to the handler without creating a scope per event. // Previously this used ScopedEventHandlerExecutor which created and disposed a scope per event // for no benefit — the handler is singleton and resolves to the same instance regardless. @@ -211,11 +237,16 @@ public static IServiceCollection AddSingletonEventHandler(this IS /// The service collection. /// If true, uses persistent subscriptions; otherwise, uses catch-up subscriptions. /// The stream position to start reading from. If null, defaults to the start of the stream. + /// Where to source the merged stream. Default (unchanged). + /// is opt-in and CATCH-UP-only, with WEAKER CaughtUp semantics — see that + /// enum value: CaughtUp may fire before all history is delivered, so do NOT opt in if the read model uses + /// ICaughtUpHandler as a "fully caught up / now authoritative" readiness signal. No-loss + commit order still hold. /// The service collection for method chaining. - public static IServiceCollection AddEventHandler(this IServiceCollection services, bool persistently = false, FromStream? start = null) where TEventHandler : class, IEventHandler, ITypeRegister + public static IServiceCollection AddEventHandler(this IServiceCollection services, bool persistently = false, FromStream? start = null, MergeSource mergeSource = MergeSource.Projection) where TEventHandler : class, IEventHandler, ITypeRegister { + ValidateMergeSource(mergeSource, persistently, start); services.AddSingleton>(); - services.AddSingleton(sp => sp.GetRequiredService>().Configure(persistently, start)); + services.AddSingleton(sp => sp.GetRequiredService>().Configure(persistently, start, mergeSource)); // ScopedEventHandlerExecutor creates a new scope per event so the handler and its // scoped dependencies (e.g. DbContext) get proper lifetime management. services.AddSingleton, ScopedEventHandlerExecutor>(); diff --git a/src/MicroPlumberd.Services/EventHandlerStarter.cs b/src/MicroPlumberd.Services/EventHandlerStarter.cs index b82da2f..810451b 100644 --- a/src/MicroPlumberd.Services/EventHandlerStarter.cs +++ b/src/MicroPlumberd.Services/EventHandlerStarter.cs @@ -12,6 +12,7 @@ class EventHandlerStarter(PlumberEngine plumber) : IEventHandlerStarte private FromStream _startPosition; private FromRelativeStreamPosition _relativeStartPosition; private bool _persistently; + private MergeSource _mergeSource; /// /// Starts the event handler subscription with the configured settings. /// @@ -21,7 +22,9 @@ public async Task Start(CancellationToken stoppingToken) { try { - if (!_persistently) + if (_mergeSource == MergeSource.UserDefinedIndex) + await plumber.SubscribeEventHandlerViaIndex(start: _relativeStartPosition, token: stoppingToken); + else if (!_persistently) await plumber.SubscribeEventHandler(start: _relativeStartPosition, token: stoppingToken); else await plumber.SubscribeEventHandlerPersistently(startFrom: _startPosition.ToStreamPosition(), token: stoppingToken); @@ -33,16 +36,19 @@ public async Task Start(CancellationToken stoppingToken) } /// - /// Configures the event handler with persistence and start position settings. + /// Configures the event handler with persistence, start position, and merge-source settings. /// /// If true, uses persistent subscriptions; otherwise, uses catch-up subscriptions. /// The stream position to start reading from. + /// Where to source the merged stream (projection default, or a user-defined index). /// This starter instance for method chaining. - public EventHandlerStarter Configure(bool persistently = false, FromStream? start = null) + public EventHandlerStarter Configure(bool persistently = false, FromStream? start = null, + MergeSource mergeSource = MergeSource.Projection) { this._persistently = persistently; this._startPosition = start ?? FromStream.Start; this._relativeStartPosition = _startPosition; + this._mergeSource = mergeSource; return this; } /// diff --git a/src/MicroPlumberd.Services/MergeSource.cs b/src/MicroPlumberd.Services/MergeSource.cs new file mode 100644 index 0000000..97f981c --- /dev/null +++ b/src/MicroPlumberd.Services/MergeSource.cs @@ -0,0 +1,33 @@ +namespace MicroPlumberd.Services; + +/// +/// Selects where a catch-up read model sources its MERGED stream from. The default is +/// (unchanged for every existing app); is an opt-in, +/// additive path that reads a KurrentDB 26.1 user-defined index via filtered $all instead of a +/// fromStreams(...).linkTo(outputStream) join projection. See +/// docs/design-index-backed-merge-streams.md. +/// +public enum MergeSource +{ + /// The default: a fromStreams(...).linkTo(outputStream) join projection (unchanged). + Projection = 0, + + /// + /// Opt-in: a KurrentDB user-defined index read via filtered $all. Catch-up subscriptions ONLY — + /// persistent subscriptions cannot see index links (SPIKE-7), so persistently: true is rejected. + /// + /// Guarantees: no-loss and commit ordering (identical to the projection path), and + /// ICaughtUpHandler.CaughtUp() DOES fire. + /// + /// + /// WEAKER CaughtUp semantics than projection-backed — read before opting in. An index-backed + /// subscription's history→live boundary tracks the $all position, so while the index is still + /// BACKFILLING, CaughtUp can fire BEFORE all historical (backfilled) links have been delivered — they + /// then arrive as "live". Unlike the projection output-stream path, CaughtUp here does NOT guarantee + /// that all history was processed first. A read model that treats ICaughtUpHandler.CaughtUp() as a + /// "fully caught up / now authoritative" readiness signal should stay on (or must + /// tolerate the weaker guarantee). Eventual delivery of all history and commit ordering are still guaranteed. + /// + /// + UserDefinedIndex = 1 +} diff --git a/src/MicroPlumberd.Tests/Integration/IndexBackedMergeReviewFollowupsTests.cs b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeReviewFollowupsTests.cs new file mode 100644 index 0000000..eb667fc --- /dev/null +++ b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeReviewFollowupsTests.cs @@ -0,0 +1,104 @@ +using FluentAssertions; +using Microsoft.Extensions.DependencyInjection; +using MicroPlumberd.Services; +using MicroPlumberd.Testing; +using MicroPlumberd.Tests.App.Domain; +using MicroPlumberd.Tests.App.Infrastructure; +using MicroPlumberd.Tests.Utils; +using Xunit.Abstractions; + +namespace MicroPlumberd.Tests.Integration; + +/// +/// Review follow-ups closed before v1: (CLOSE #1) the full-DI ACCEPTANCE PATH a real consumer wires — host boot → +/// EventHandlerStarter → SubscribeEventHandlerViaIndex(eh:null) DI resolution — and (CLOSE #3) the cross-output +/// managed-index-name COLLISION guard (a punctuation collision must fail loud, not silently DELETE the wrong index). +/// +public class IndexBackedMergeReviewFollowupsTests +{ + // CLOSE #3 — collision guard (no server): two DISTINCT output streams that normalize to the same managed base + // must be rejected; the SAME output stream re-registering is idempotent. + [Fact] + public void ManagedBase_collision_is_rejected_same_name_is_idempotent() + { + UserDefinedIndex.RegisterManagedBase("CollFoo.1"); // claims base "collfoo-1" + var idempotent = () => UserDefinedIndex.RegisterManagedBase("CollFoo.1"); + idempotent.Should().NotThrow("re-registering the SAME output stream must be a no-op"); + + var collision = () => UserDefinedIndex.RegisterManagedBase("CollFoo-1"); // also normalizes to "collfoo-1" + collision.Should().Throw() + .WithMessage("*collision*") + .Which.Message.Should().Contain("collfoo-1"); + } + + // CLOSE #1 — the full DI acceptance path (real KurrentDB 26.1). A consumer registers the handler with + // MergeSource.UserDefinedIndex and starts the host; EventHandlerService → EventHandlerStarter.Start → + // SubscribeEventHandlerViaIndex(eh:null) resolves the handler from DI and folds index-delivered events. This + // exercises the eh==null DI-resolution + starter routing that the instance-passing tests bypass. + [TestCategory("Integration")] + public class DiAcceptancePath(ITestOutputHelper output) + { + private static string? NameOf(object? payload) => payload switch + { + FooCreated c => c.Name, + FooRefined r => r.Name, + _ => null + }; + + private static async Task WaitUntil(Func condition, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + if (condition()) return true; + await Task.Delay(200); + } + return condition(); + } + + [Fact] + public async Task Full_di_index_backed_handler_folds_events_and_tails_live() + { + var es = EventStoreServer.Create($"mp-di-{Guid.NewGuid():N}"); + await es.StartInDocker(inMemory: true); + TestAppHost? host = null; + try + { + var settings = es.GetEventStoreSettings(); + var appender = Plumber.Create(settings); + + // History appended before the host boots. + for (var i = 0; i < 3; i++) await appender.SaveNew(FooAggregate.Open(i.ToString(), Guid.NewGuid())); + + // The FULL DI path a real consumer uses: opt in via mergeSource, boot the host (blocks until the + // starter has subscribed via SubscribeEventHandlerViaIndex(eh:null)). + host = new TestAppHost(output); + host.Configure(x => x + .AddPlumberd(settings) + .AddSingleton() + .AddSingletonEventHandler(mergeSource: MergeSource.UserDefinedIndex)); + var sp = await host.StartAsync(); + + var model = sp.GetRequiredService(); // the DI-resolved singleton the runner drives + (await WaitUntil(() => model.AssertionDb.Index.Count >= 3, TimeSpan.FromSeconds(60))) + .Should().BeTrue("the DI-wired index-backed handler must fold the 3 historical events"); + + DeliveredOrder(model).Should().Equal(new[] { "0", "1", "2" }, "history folded in commit order via the index"); + + // Live after boot, still through the DI-resolved handler. + await appender.SaveNew(FooAggregate.Open("3", Guid.NewGuid())); + (await WaitUntil(() => model.AssertionDb.Index.Count >= 4, TimeSpan.FromSeconds(30))) + .Should().BeTrue("post-boot appends tail into the DI-wired index-backed handler"); + DeliveredOrder(model).Should().Equal(new[] { "0", "1", "2", "3" }); + } + finally + { + host?.Dispose(); + await es.DisposeAsync(); + } + } + + private static string[] DeliveredOrder(CaughtUpFooModel m) => + m.Timeline.Where(t => t.Kind == "event").Select(t => NameOf(t.Payload)).ToArray()!; + } +} diff --git a/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS1Tests.cs b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS1Tests.cs new file mode 100644 index 0000000..cb57fe3 --- /dev/null +++ b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS1Tests.cs @@ -0,0 +1,237 @@ +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Testing; +using MicroPlumberd.Tests.App.Domain; +using MicroPlumberd.Tests.App.Infrastructure; +using MicroPlumberd.Tests.Utils; + +namespace MicroPlumberd.Tests.Integration; + +/// +/// S1 — the index-backed subscription PRIMITIVE + the refactored seam, against a +/// REAL KurrentDB 26.1 ( Docker, NEVER Testcontainers). Exercises the SHIPPED types +/// (, , the SubscriptionRunnerState guard) +/// through real MicroPlumberd events (FooCreated/FooRefined) — not ad-hoc client calls. +/// +/// S1-T1 catch-up→live in commit order; S1-T2 resume from a recorded $all position; S1-T3 the SubscribeToStream +/// footgun guard; S1-T4 the stream-backed path still delivers (the refactor changed the seam, not the behaviour). +/// (S1-T5 Migration parity is the pre-existing UserDefinedIndexIntegrationTests, unchanged by the relocation.) +/// +[TestCategory("Integration")] +public class IndexBackedMergeS1Tests +{ + private sealed class Fixture : IAsyncDisposable + { + private readonly List _servers = new(); + + public async Task<(KurrentDBClientSettings Settings, PlumberEngine Engine, IPlumber Plumber)> NewAsync(string tag) + { + var es = EventStoreServer.Create($"mp-s1-{tag}-{Guid.NewGuid():N}"); + _servers.Add(es); + await es.StartInDocker(inMemory: true); + var settings = es.GetEventStoreSettings(); + return (settings, new PlumberEngine(settings), Plumber.Create(settings)); + } + + public async ValueTask DisposeAsync() + { + foreach (var s in _servers) await s.DisposeAsync(); + } + } + + private static string? NameOf(object? payload) => payload switch + { + FooCreated c => c.Name, + FooRefined r => r.Name, + _ => null + }; + + private static async Task AppendFoo(IPlumber plumber, string name) => + await plumber.SaveNew(FooAggregate.Open(name, Guid.NewGuid())); + + private static async Task WaitUntil(Func condition, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + if (condition()) return true; + await Task.Delay(200); + } + return condition(); + } + + // Count materialised index links via the filtered-$all read, tolerating the "still building" NotFound + // (the index read stream 404s until the backfill has written links). Test scaffolding only. + private static async Task IndexLinkCount(KurrentDBClient client, string indexStream) + { + var read = client.ReadAllAsync(Direction.Forwards, Position.Start, StreamFilter.Prefix(indexStream), + maxCount: long.MaxValue, resolveLinkTos: true); + var e = read.GetAsyncEnumerator(); + var n = 0; + try + { + while (true) + { + try { if (!await e.MoveNextAsync()) break; } + catch (Grpc.Core.RpcException ex) when (ex.StatusCode == Grpc.Core.StatusCode.NotFound) { return 0; } + if (e.Current.Event is not null) n++; + } + } + finally { await e.DisposeAsync(); } + return n; + } + + private async Task<(string Name, string IndexStream)> EnsureIndexForFooModelAsync(PlumberEngine engine) + { + var outputStream = engine.Conventions.OutputStreamModelConvention(typeof(CaughtUpFooModel)); + var types = engine.TypeHandlerRegisters.GetEventNamesFor().ToHashSet(StringComparer.Ordinal); + var filter = UserDefinedIndex.BuildFilter(types); + var name = UserDefinedIndex.IndexNameFor(outputStream, filter); + await engine.UserDefinedIndex.EnsureAsync(name, types); + return (name, UserDefinedIndex.IndexStream(name)); + } + + // S1-T1 — index-backed subscription catches up over history THEN tails live appends, in commit order, + // driven by IndexSubscriptionState inside a SubscriptionRunner (the shared loop), through production types. + [Fact] + public async Task S1T1_index_backed_subscription_catches_up_then_tails_in_commit_order() + { + await using var fx = new Fixture(); + var (_, engine, plumber) = await fx.NewAsync("t1"); + + // History: three FooCreated events named "0","1","2" (three streams → interleaved in $all commit order). + for (var i = 0; i < 3; i++) await AppendFoo(plumber, i.ToString()); + + var (_, indexStream) = await EnsureIndexForFooModelAsync(engine); + + var model = new CaughtUpFooModel(new InMemoryAssertionDb()); + using var cts = new CancellationTokenSource(); + var state = new IndexSubscriptionState(engine.Client, indexStream, FromAll.Start, null, cts.Token); + var runner = new SubscriptionRunner(engine, state); + await runner.WithHandler(model, engine.TypeHandlerRegisters.GetEventNameConverterFor()!); + + (await WaitUntil(() => model.AssertionDb.Index.Count >= 3, TimeSpan.FromSeconds(30))) + .Should().BeTrue("the index-backed subscription must catch up over the 3 historical events"); + + // Live: append two more AFTER the subscription is open; the SAME subscription must tail them. + await AppendFoo(plumber, "3"); + await AppendFoo(plumber, "4"); + + (await WaitUntil(() => model.AssertionDb.Index.Count >= 5, TimeSpan.FromSeconds(30))) + .Should().BeTrue("post-subscription appends must be pushed to the open filtered-$all subscription"); + + await cts.CancelAsync(); + await runner.DisposeAsync(); + + var delivered = model.Timeline.Where(t => t.Kind == "event").Select(t => NameOf(t.Payload)).ToArray(); + delivered.Should().Equal(new[] { "0", "1", "2", "3", "4" }, + "history + live must arrive in $all commit order via the index"); + } + + // S1-T2 — resume: record the last $all position after 0,1,2, dispose, append 3,4, resubscribe from the + // recorded FromAll position → ONLY 3,4 are delivered (no replay of 0,1,2). Mirrors SPIKE-3. + [Fact] + public async Task S1T2_resume_from_recorded_position_delivers_only_new_events() + { + await using var fx = new Fixture(); + var (_, engine, plumber) = await fx.NewAsync("t2"); + + for (var i = 0; i < 3; i++) await AppendFoo(plumber, i.ToString()); + var (_, indexStream) = await EnsureIndexForFooModelAsync(engine); + var filter = new SubscriptionFilterOptions(StreamFilter.Prefix(indexStream)); + + // Deterministic scaffolding: wait for the index to materialise its 3 history links before pass 1 opens a + // raw (non-retrying) subscription. The SHIPPED path (IndexSubscriptionState in SubscriptionRunner) instead + // tolerates the transient "still building" NotFound via resubscribe — exercised in S1T1. + var indexReady = false; + for (var deadline = DateTime.UtcNow + TimeSpan.FromSeconds(30); DateTime.UtcNow < deadline; await Task.Delay(200)) + if (await IndexLinkCount(engine.Client, indexStream) >= 3) { indexReady = true; break; } + indexReady.Should().BeTrue("the index must backfill its 3 history links before the checkpoint pass"); + + // Pass 1: consume 0,1,2 and record the last delivered link's $all position. + var firstPass = new List(); + Position? checkpoint = null; + using (var cts1 = new CancellationTokenSource()) + { + var pump = Task.Run(async () => + { + await using var sub = engine.Client.SubscribeToAll(FromAll.Start, resolveLinkTos: true, + filterOptions: filter, cancellationToken: cts1.Token); + try + { + await foreach (var m in sub.Messages.WithCancellation(cts1.Token)) + if (m is StreamMessage.Event(var e)) + { + firstPass.Add(System.Text.Json.Nodes.JsonNode.Parse(e.Event.Data.Span)?["Name"]?.GetValue()); + if (e.OriginalPosition is { } p) checkpoint = p; + } + } + catch (OperationCanceledException) { } + }); + await WaitUntil(() => firstPass.Count >= 3, TimeSpan.FromSeconds(30)); + await cts1.CancelAsync(); + await pump; + } + + firstPass.Should().Equal(new[] { "0", "1", "2" }); + checkpoint.Should().NotBeNull("a resumable $all position must be recoverable from the index link event"); + + await AppendFoo(plumber, "3"); + await AppendFoo(plumber, "4"); + + // Pass 2: resume via IndexSubscriptionState constructed at FromAll.After(checkpoint) — the shipped type. + var model = new CaughtUpFooModel(new InMemoryAssertionDb()); + using var cts2 = new CancellationTokenSource(); + var state = new IndexSubscriptionState(engine.Client, indexStream, FromAll.After(checkpoint!.Value), null, cts2.Token); + var runner = new SubscriptionRunner(engine, state); + await runner.WithHandler(model, engine.TypeHandlerRegisters.GetEventNameConverterFor()!); + + (await WaitUntil(() => model.AssertionDb.Index.Count >= 2, TimeSpan.FromSeconds(30))) + .Should().BeTrue("resume must deliver the two events after the checkpoint"); + await Task.Delay(1000); // give any erroneous replay of 0,1,2 a chance to show up + + await cts2.CancelAsync(); + await runner.DisposeAsync(); + + var delivered = model.Timeline.Where(t => t.Kind == "event").Select(t => NameOf(t.Payload)).ToArray(); + delivered.Should().Equal(new[] { "3", "4" }, + "resuming from the recorded position yields ONLY events after it — no replay of 0,1,2"); + } + + // S1-T3 — the loud guard: SubscribeToStream against a $idx-user-… name is the SPIKE-2a/SPIKE-9 silent-zero + // footgun; StreamSubscriptionState.Subscribe() must throw InvalidOperationException instead of hanging silent. + [Fact] + public void S1T3_subscribe_to_index_stream_via_stream_state_throws() + { + using var client = new KurrentDBClient(KurrentDBClientSettings.Create("esdb://localhost:2113?tls=false")); + var state = new SubscriptionRunnerState(FromStream.Start, client, + UserDefinedIndex.IndexStream("some-merge"), null, CancellationToken.None); + + var act = () => state.Subscribe(); + act.Should().Throw() + .WithMessage("*index*") + .Which.Message.Should().Contain("SubscribeToAll"); + } + + // S1-T4 — no regression: the stream-backed (projection output-stream) path still delivers in order after the + // ISubscriptionState refactor (control — the seam changed, not the behaviour). + [Fact] + public async Task S1T4_stream_backed_subscription_still_delivers_in_order() + { + await using var fx = new Fixture(); + var (_, _, plumber) = await fx.NewAsync("t4"); + + await plumber.TryCreateJoinProjection(); + for (var i = 0; i < 3; i++) await AppendFoo(plumber, i.ToString()); + await Task.Delay(2000); // let the projection link the 3 events + + var model = new CaughtUpFooModel(new InMemoryAssertionDb()); + await plumber.SubscribeEventHandler(model, ensureOutputStreamProjection: false); + + (await WaitUntil(() => model.AssertionDb.Index.Count >= 3, TimeSpan.FromSeconds(30))) + .Should().BeTrue("the stream-backed path must still catch up over history"); + + var delivered = model.Timeline.Where(t => t.Kind == "event").Select(t => NameOf(t.Payload)).ToArray(); + delivered.Should().Equal(new[] { "0", "1", "2" }, "stream-backed delivery order is unchanged by the refactor"); + } +} diff --git a/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS2Tests.cs b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS2Tests.cs new file mode 100644 index 0000000..b8b5c9c --- /dev/null +++ b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS2Tests.cs @@ -0,0 +1,184 @@ +using FluentAssertions; +using KurrentDB.Client; +using Microsoft.Extensions.DependencyInjection; +using MicroPlumberd.Services; +using MicroPlumberd.Testing; +using MicroPlumberd.Tests.App.Domain; +using MicroPlumberd.Tests.App.Infrastructure; +using MicroPlumberd.Tests.Utils; + +namespace MicroPlumberd.Tests.Integration; + +/// +/// S2 — opt-in wiring + the load-bearing PROJECTION-vs-INDEX PARITY test against real KurrentDB 26.1. +/// +/// S2-T1 (acceptance bar): the SAME read model fed the SAME interleaved events via the projection path and the +/// index path reaches IDENTICAL final state in the SAME delivery order. S2-T2 live-after-boot; S2-T3 +/// ICaughtUpHandler fires on the index path; S2-T4 registration guards (persistent / revision-start rejected). +/// +[TestCategory("Integration")] +public class IndexBackedMergeS2Tests +{ + private sealed class Fixture : IAsyncDisposable + { + private readonly List _servers = new(); + + public async Task<(PlumberEngine Engine, IPlumber Plumber)> NewAsync(string tag) + { + var es = EventStoreServer.Create($"mp-s2-{tag}-{Guid.NewGuid():N}"); + _servers.Add(es); + await es.StartInDocker(inMemory: true); + var settings = es.GetEventStoreSettings(); + return (new PlumberEngine(settings), Plumber.Create(settings)); + } + + public async ValueTask DisposeAsync() + { + foreach (var s in _servers) await s.DisposeAsync(); + } + } + + private static string? NameOf(object? payload) => payload switch + { + FooCreated c => c.Name, + FooRefined r => r.Name, + _ => null + }; + + private static string[] DeliveredOrder(CaughtUpFooModel m) => + m.Timeline.Where(t => t.Kind == "event").Select(t => NameOf(t.Payload)).ToArray()!; + + private static async Task WaitUntil(Func condition, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + if (condition()) return true; + await Task.Delay(200); + } + return condition(); + } + + // Appends an interleaved FooCreated/FooRefined sequence across three aggregates; the Names are "0".."4" in + // $all commit (call) order — the exact merged order BOTH paths must reproduce. + private static async Task AppendInterleavedAsync(IPlumber plumber) + { + var a = Guid.NewGuid(); + var b = Guid.NewGuid(); + var c = Guid.NewGuid(); + await plumber.SaveNew(FooAggregate.Open("0", a)); // FooCreated 0 + await plumber.SaveNew(FooAggregate.Open("1", b)); // FooCreated 1 + var aa = await plumber.Get(a); aa.Refine("2"); await plumber.SaveChanges(aa); // FooRefined 2 + await plumber.SaveNew(FooAggregate.Open("3", c)); // FooCreated 3 + var bb = await plumber.Get(b); bb.Refine("4"); await plumber.SaveChanges(bb); // FooRefined 4 + return new[] { "0", "1", "2", "3", "4" }; + } + + // S2-T1 — THE acceptance bar. Same events, two sources (projection + index), identical final state & order. + [Fact] + public async Task S2T1_projection_and_index_paths_produce_identical_state_and_order() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t1"); + + var expected = await AppendInterleavedAsync(plumber); + + var viaProjection = new CaughtUpFooModel(new InMemoryAssertionDb()); + var viaIndex = new CaughtUpFooModel(new InMemoryAssertionDb()); + + // Projection path (the unchanged default) and index path (opt-in) — SAME events, SAME handler type. + await engine.SubscribeEventHandler(viaProjection); + await engine.SubscribeEventHandlerViaIndex(viaIndex); + + (await WaitUntil(() => viaProjection.AssertionDb.Index.Count >= 5 && viaIndex.AssertionDb.Index.Count >= 5, + TimeSpan.FromSeconds(45))).Should().BeTrue("both paths must catch up over all 5 events"); + + var projectionOrder = DeliveredOrder(viaProjection); + var indexOrder = DeliveredOrder(viaIndex); + + projectionOrder.Should().Equal(expected, "the projection path delivers the merge in commit order"); + indexOrder.Should().Equal(expected, "the index path delivers the merge in the SAME commit order"); + indexOrder.Should().Equal(projectionOrder, + "PARITY: index-backed merge gives exactly the ordering + no-loss + no-double-process the projection path gives"); + viaIndex.CaughtUpCount.Should().BeGreaterThanOrEqualTo(1, "the index path fires ICaughtUpHandler after history"); + } + + // S2-T2 — live after boot: with the index-backed handler running, post-boot appends are delivered live, in + // commit order (SPIKE-1/2b end-to-end through the framework). + [Fact] + public async Task S2T2_index_backed_handler_tails_live_appends_after_boot() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t2"); + + await plumber.SaveNew(FooAggregate.Open("0", Guid.NewGuid())); + + var model = new CaughtUpFooModel(new InMemoryAssertionDb()); + await engine.SubscribeEventHandlerViaIndex(model); + + (await WaitUntil(() => model.IsLive, TimeSpan.FromSeconds(45))) + .Should().BeTrue("the index-backed handler must catch up and go live"); + + await plumber.SaveNew(FooAggregate.Open("1", Guid.NewGuid())); + await plumber.SaveNew(FooAggregate.Open("2", Guid.NewGuid())); + + (await WaitUntil(() => model.AssertionDb.Index.Count >= 3, TimeSpan.FromSeconds(30))) + .Should().BeTrue("post-boot appends must tail into the index-backed subscription"); + + DeliveredOrder(model).Should().Equal(new[] { "0", "1", "2" }, "live appends arrive in commit order"); + } + + // S2-T3 — ICaughtUpHandler is wired on the index path (regression-guards OPEN-4): CaughtUp fires (read model + // marks itself ready/live), then live appends still tail in commit order. NOTE: unlike the projection + // output-stream path, an index-backed subscription's history→live boundary tracks the $all position, so while + // the index is still backfilling CaughtUp can fire before all backfilled links arrive — the guaranteed + // properties are "CaughtUp fires" + "no loss" + "commit order", not a strict history-before-CaughtUp timeline. + [Fact] + public async Task S2T3_caughtup_handler_fires_and_live_tail_preserves_order_on_index_path() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t3"); + + await plumber.SaveNew(FooAggregate.Open("0", Guid.NewGuid())); + await plumber.SaveNew(FooAggregate.Open("1", Guid.NewGuid())); + + var model = new CaughtUpFooModel(new InMemoryAssertionDb()); + await engine.SubscribeEventHandlerViaIndex(model); + + (await WaitUntil(() => model.CaughtUpCount >= 1, TimeSpan.FromSeconds(45))) + .Should().BeTrue("CaughtUp must fire so the read model can mark itself ready"); + model.IsLive.Should().BeTrue("CaughtUp sets IsLive"); + + (await WaitUntil(() => model.AssertionDb.Index.Count >= 2, TimeSpan.FromSeconds(30))) + .Should().BeTrue("both historical events must be delivered (no loss)"); + + await plumber.SaveNew(FooAggregate.Open("2", Guid.NewGuid())); + (await WaitUntil(() => model.AssertionDb.Index.Count >= 3, TimeSpan.FromSeconds(30))) + .Should().BeTrue("the post-CaughtUp live append must tail in"); + + DeliveredOrder(model).Should().Equal(new[] { "0", "1", "2" }, "history + live preserved in commit order"); + } + + // S2-T4 — registration guards (fail fast, no server needed): UserDefinedIndex is catch-up-only and Start/End-only. + [Fact] + public void S2T4_registration_guards_reject_persistent_and_revision_start() + { + var persistent = () => new ServiceCollection() + .AddSingletonEventHandler(persistently: true, mergeSource: MergeSource.UserDefinedIndex); + persistent.Should().Throw() + .WithMessage("*catch-up*", "persistent + index-backed must be rejected at registration (SPIKE-7)"); + + var revisionStart = () => new ServiceCollection() + .AddSingletonEventHandler(start: FromStream.After(5), mergeSource: MergeSource.UserDefinedIndex); + revisionStart.Should().Throw() + .WithMessage("*Start or End*", "a specific revision start is meaningless on filtered $all"); + + // Start and End are allowed. + var startOk = () => new ServiceCollection() + .AddSingletonEventHandler(start: FromStream.Start, mergeSource: MergeSource.UserDefinedIndex); + var endOk = () => new ServiceCollection() + .AddSingletonEventHandler(start: FromStream.End, mergeSource: MergeSource.UserDefinedIndex); + startOk.Should().NotThrow(); + endOk.Should().NotThrow(); + } +} diff --git a/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS3Tests.cs b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS3Tests.cs new file mode 100644 index 0000000..76e1ca8 --- /dev/null +++ b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS3Tests.cs @@ -0,0 +1,169 @@ +using FluentAssertions; +using KurrentDB.Client; +using MicroPlumberd.Testing; +using MicroPlumberd.Tests.App.Domain; +using MicroPlumberd.Tests.Utils; + +namespace MicroPlumberd.Tests.Integration; + +/// +/// S3 — create-new-and-swap reconciler () against real KurrentDB 26.1. +/// In-place index redefine is impossible (SPIKE-5/409), so a changed event-type SET yields a NEW filter-hashed +/// index name (mpidx-<output>-<hash8>), the read model rebuilds from the new merged view, and the +/// superseded index is DELETEd. +/// +/// S3-T1 definition change → new index + rebuilt (widened) content; S3-T2 no-op when the type set is unchanged; +/// S3-T3 orphan cleanup — the old-hash index is DELETEd and no longer resolves. +/// +[TestCategory("Integration")] +public class IndexBackedMergeS3Tests +{ + private sealed class Fixture : IAsyncDisposable + { + private readonly List _servers = new(); + + public async Task<(PlumberEngine Engine, IPlumber Plumber)> NewAsync(string tag) + { + var es = EventStoreServer.Create($"mp-s3-{tag}-{Guid.NewGuid():N}"); + _servers.Add(es); + await es.StartInDocker(inMemory: true); + var settings = es.GetEventStoreSettings(); + return (new PlumberEngine(settings), Plumber.Create(settings)); + } + + public async ValueTask DisposeAsync() + { + foreach (var s in _servers) await s.DisposeAsync(); + } + } + + private const string OutputStream = "S3Merge"; + + private static (IReadOnlySet Created, IReadOnlySet CreatedAndRefined) TypeSets(PlumberEngine engine) + { + var all = engine.TypeHandlerRegisters.GetEventNamesFor().ToHashSet(StringComparer.Ordinal); + var createdOnly = all.Where(n => n.Contains("Created")).ToHashSet(StringComparer.Ordinal); + return (createdOnly, all); + } + + private static async Task AppendCorpusAsync(IPlumber plumber) + { + var a = Guid.NewGuid(); + var b = Guid.NewGuid(); + await plumber.SaveNew(FooAggregate.Open("0", a)); // FooCreated 0 + await plumber.SaveNew(FooAggregate.Open("1", b)); // FooCreated 1 + var aa = await plumber.Get(a); aa.Refine("2"); await plumber.SaveChanges(aa); // FooRefined 2 + } + + private static async Task> ReadIndexNamesAsync(KurrentDBClient client, string indexStream) + { + var names = new List(); + var read = client.ReadAllAsync(Direction.Forwards, Position.Start, StreamFilter.Prefix(indexStream), + maxCount: long.MaxValue, resolveLinkTos: true); + var e = read.GetAsyncEnumerator(); + try + { + while (true) + { + try { if (!await e.MoveNextAsync()) break; } + catch (Grpc.Core.RpcException ex) when (ex.StatusCode == Grpc.Core.StatusCode.NotFound) { break; } + if (e.Current.Event is null) continue; + names.Add(System.Text.Json.Nodes.JsonNode.Parse(e.Current.Event.Data.Span)?["Name"]?.GetValue()); + } + } + finally { await e.DisposeAsync(); } + return names; + } + + private static async Task> WaitIndexAsync(KurrentDBClient client, string indexStream, int atLeast) + { + var names = new List(); + for (var deadline = DateTime.UtcNow + TimeSpan.FromSeconds(30); DateTime.UtcNow < deadline; await Task.Delay(200)) + { + names = await ReadIndexNamesAsync(client, indexStream); + if (names.Count >= atLeast) return names; + } + return names; + } + + // S3-T1 — a definition change (widen {Created} → {Created,Refined}) creates a NEW filter-hashed index; the + // new index includes the Refined event; the old index is superseded. + [Fact] + public async Task S3T1_definition_change_creates_new_index_with_widened_content() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t1"); + await AppendCorpusAsync(plumber); + + var (created, all) = TypeSets(engine); + var reconciler = new UserDefinedIndexReconciler(engine.UserDefinedIndex); + + var name1 = await reconciler.ReconcileAsync(OutputStream, created, default); + name1.Should().Be(UserDefinedIndex.IndexNameFor(OutputStream, UserDefinedIndex.BuildFilter(created))); + var v1 = await WaitIndexAsync(engine.Client, UserDefinedIndex.IndexStream(name1), 2); + v1.Should().Equal(new[] { "0", "1" }, "the {Created}-only index merges only FooCreated, in commit order"); + + var name2 = await reconciler.ReconcileAsync(OutputStream, all, default); + name2.Should().NotBe(name1, "a changed event-type set deterministically yields a NEW filter-hashed name"); + name2.Should().Be(UserDefinedIndex.IndexNameFor(OutputStream, UserDefinedIndex.BuildFilter(all))); + var v2 = await WaitIndexAsync(engine.Client, UserDefinedIndex.IndexStream(name2), 3); + v2.Should().Equal(new[] { "0", "1", "2" }, "the widened index rebuilds to include FooRefined, in commit order"); + } + + // S3-T2 — no-op when the type set is unchanged: same name, 409 reuse, no delete of the current index. + [Fact] + public async Task S3T2_reconcile_is_noop_when_type_set_unchanged() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t2"); + await AppendCorpusAsync(plumber); + + var (_, all) = TypeSets(engine); + var reconciler = new UserDefinedIndexReconciler(engine.UserDefinedIndex); + + var first = await reconciler.ReconcileAsync(OutputStream, all, default); + await WaitIndexAsync(engine.Client, UserDefinedIndex.IndexStream(first), 3); + + var second = await reconciler.ReconcileAsync(OutputStream, all, default); + second.Should().Be(first, "an unchanged type set reconciles to the SAME index name (409 reuse)"); + + (await engine.UserDefinedIndex.GetFilterAsync(first)).Should() + .NotBeNull("the current index must NOT be deleted by an unchanged reconcile"); + (await ReadIndexNamesAsync(engine.Client, UserDefinedIndex.IndexStream(first))) + .Should().Equal(new[] { "0", "1", "2" }, "the index is untouched by an idempotent reconcile"); + } + + // S3-T3 — orphan cleanup: after a swap, the old-hash index is DELETEd (no longer resolves) and only the + // current-hash index remains among the managed indexes for this output stream. + [Fact] + public async Task S3T3_superseded_index_is_deleted_after_swap() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t3"); + await AppendCorpusAsync(plumber); + + var (created, all) = TypeSets(engine); + var reconciler = new UserDefinedIndexReconciler(engine.UserDefinedIndex); + + var oldName = await reconciler.ReconcileAsync(OutputStream, created, default); + await WaitIndexAsync(engine.Client, UserDefinedIndex.IndexStream(oldName), 2); + (await engine.UserDefinedIndex.GetFilterAsync(oldName)).Should().NotBeNull("the old index exists before the swap"); + + var newName = await reconciler.ReconcileAsync(OutputStream, all, default); + newName.Should().NotBe(oldName); + + // The reconcile deletes the superseded index. Give the DELETE a moment to take effect, then confirm it is + // gone and the current index remains. + var deleted = false; + for (var deadline = DateTime.UtcNow + TimeSpan.FromSeconds(15); DateTime.UtcNow < deadline; await Task.Delay(300)) + if (await engine.UserDefinedIndex.GetFilterAsync(oldName) is null) { deleted = true; break; } + + deleted.Should().BeTrue("the superseded old-hash index must be DELETEd by the reconciler"); + (await engine.UserDefinedIndex.GetFilterAsync(newName)).Should().NotBeNull("the current index must remain"); + + var prefix = UserDefinedIndex.ManagedNamePrefixFor(OutputStream); + var managed = (await engine.UserDefinedIndex.ListNamesAsync()) + .Where(n => n.StartsWith(prefix, StringComparison.Ordinal)).ToArray(); + managed.Should().OnlyContain(n => n == newName, "only the current-hash managed index should remain"); + } +} diff --git a/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS4Tests.cs b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS4Tests.cs new file mode 100644 index 0000000..4bf2f04 --- /dev/null +++ b/src/MicroPlumberd.Tests/Integration/IndexBackedMergeS4Tests.cs @@ -0,0 +1,161 @@ +using FluentAssertions; +using KurrentDB.Client; +using Microsoft.Extensions.DependencyInjection; +using MicroPlumberd.Services; +using MicroPlumberd.Testing; +using MicroPlumberd.Tests.App.Domain; +using MicroPlumberd.Tests.App.Infrastructure; +using MicroPlumberd.Tests.Utils; + +namespace MicroPlumberd.Tests.Integration; + +/// +/// S4 — coexistence + the PERMANENT persistent-subscription guard + the default staying unchanged + proliferation +/// headroom, against real KurrentDB 26.1. +/// +/// S4-T1 a projection-backed and an index-backed handler run side by side, both correct; S4-T2 a handler with no +/// mergeSource stays projection-backed and creates NO index; S4-T3 UserDefinedIndex+persistently is rejected +/// (permanent, SPIKE-7) while a normal persistent registration is untouched; S4-T4 many index-backed handlers in +/// one app all build (SPIKE-6 headroom regression guard). +/// +[TestCategory("Integration")] +public class IndexBackedMergeS4Tests +{ + private sealed class Fixture : IAsyncDisposable + { + private readonly List _servers = new(); + + public async Task<(PlumberEngine Engine, IPlumber Plumber)> NewAsync(string tag) + { + var es = EventStoreServer.Create($"mp-s4-{tag}-{Guid.NewGuid():N}"); + _servers.Add(es); + await es.StartInDocker(inMemory: true); + var settings = es.GetEventStoreSettings(); + return (new PlumberEngine(settings), Plumber.Create(settings)); + } + + public async ValueTask DisposeAsync() + { + foreach (var s in _servers) await s.DisposeAsync(); + } + } + + private static async Task AppendCorpusAsync(IPlumber plumber) + { + var a = Guid.NewGuid(); + var b = Guid.NewGuid(); + await plumber.SaveNew(FooAggregate.Open("0", a)); + await plumber.SaveNew(FooAggregate.Open("1", b)); + var aa = await plumber.Get(a); aa.Refine("2"); await plumber.SaveChanges(aa); + } + + private static async Task WaitUntil(Func condition, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (DateTime.UtcNow < deadline) + { + if (condition()) return true; + await Task.Delay(200); + } + return condition(); + } + + // S4-T1 — coexistence: a projection-backed handler and an index-backed handler build correct read models in + // the SAME engine (the two mechanisms run side by side). + [Fact] + public async Task S4T1_projection_and_index_handlers_coexist() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t1"); + await AppendCorpusAsync(plumber); + + var projectionBacked = new FooModel(new InMemoryAssertionDb()); + var indexBacked = new CaughtUpFooModel(new InMemoryAssertionDb()); + + await engine.SubscribeEventHandler(projectionBacked); + await engine.SubscribeEventHandlerViaIndex(indexBacked); + + (await WaitUntil(() => projectionBacked.AssertionDb.Index.Count >= 3 && indexBacked.AssertionDb.Index.Count >= 3, + TimeSpan.FromSeconds(45))).Should().BeTrue("both the projection-backed and index-backed handlers build correctly, side by side"); + } + + // S4-T2 — default unchanged: a handler subscribed with no mergeSource creates a join projection and NO managed + // index (assert mpidx--* is absent for it). + [Fact] + public async Task S4T2_default_projection_handler_creates_no_index() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t2"); + await AppendCorpusAsync(plumber); + + var model = new FooModel(new InMemoryAssertionDb()); + await engine.SubscribeEventHandler(model); // default = projection-backed + + (await WaitUntil(() => model.AssertionDb.Index.Count >= 3, TimeSpan.FromSeconds(45))) + .Should().BeTrue("the default projection path still builds the read model"); + + var outputStream = engine.Conventions.OutputStreamModelConvention(typeof(FooModel)); + var prefix = UserDefinedIndex.ManagedNamePrefixFor(outputStream); + var managed = (await engine.UserDefinedIndex.ListNamesAsync()) + .Where(n => n.StartsWith(prefix, StringComparison.Ordinal)).ToArray(); + managed.Should().BeEmpty("the projection-backed default must NOT create a user-defined index"); + } + + // S4-T3 — the PERMANENT persistent guard: UserDefinedIndex+persistently is rejected at registration (SPIKE-7), + // while a normal persistent (projection-backed) registration is untouched. Fail-fast, no server needed. + [Fact] + public void S4T3_persistent_index_rejected_projection_persistent_untouched() + { + var indexPersistent = () => new ServiceCollection() + .AddSingletonEventHandler(persistently: true, mergeSource: MergeSource.UserDefinedIndex); + indexPersistent.Should().Throw("persistent index-backing is a permanent exclusion (SPIKE-7)"); + + var projectionPersistent = () => new ServiceCollection() + .AddSingletonEventHandler(persistently: true); + projectionPersistent.Should().NotThrow("persistent projection-backed handlers are unchanged"); + } + + // S4-T4 — proliferation headroom (SPIKE-6 regression guard): many index-backed merges in one app all build. + [Fact] + public async Task S4T4_many_index_backed_merges_all_build() + { + await using var fx = new Fixture(); + var (engine, plumber) = await fx.NewAsync("t4"); + await AppendCorpusAsync(plumber); + + var types = engine.TypeHandlerRegisters.GetEventNamesFor().ToHashSet(StringComparer.Ordinal); + var reconciler = new UserDefinedIndexReconciler(engine.UserDefinedIndex); + + const int n = 6; + var indexStreams = new List(); + for (var i = 0; i < n; i++) + { + var name = await reconciler.ReconcileAsync($"S4Prolif{i}", types, default); + indexStreams.Add(UserDefinedIndex.IndexStream(name)); + } + + foreach (var stream in indexStreams) + { + var built = false; + for (var deadline = DateTime.UtcNow + TimeSpan.FromSeconds(30); DateTime.UtcNow < deadline; await Task.Delay(200)) + { + var read = engine.Client.ReadAllAsync(Direction.Forwards, Position.Start, StreamFilter.Prefix(stream), + maxCount: long.MaxValue, resolveLinkTos: true); + var count = 0; + var e = read.GetAsyncEnumerator(); + try + { + while (true) + { + try { if (!await e.MoveNextAsync()) break; } + catch (Grpc.Core.RpcException ex) when (ex.StatusCode == Grpc.Core.StatusCode.NotFound) { break; } + if (e.Current.Event is not null) count++; + } + } + finally { await e.DisposeAsync(); } + if (count >= 3) { built = true; break; } + } + built.Should().BeTrue($"index {stream} must build the full merge — {n} concurrent index-backed merges all build"); + } + } +} diff --git a/src/MicroPlumberd.sln b/src/MicroPlumberd.sln index 9125b1f..fb184da 100644 --- a/src/MicroPlumberd.sln +++ b/src/MicroPlumberd.sln @@ -73,6 +73,18 @@ Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "MicroPlumberd.Services.Uniq EndProject Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "MicroPlumberd.Services.Uniqueness.Tests", "MicroPlumberd.Services.Uniqueness.Tests\MicroPlumberd.Services.Uniqueness.Tests.csproj", "{0F09FC14-E906-4FF0-812B-25DF6B2C0D36}" EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "MicroPlumberd.Migration", "MicroPlumberd.Migration\MicroPlumberd.Migration.csproj", "{83CEAEE1-EE88-4D5C-B938-97D8123C51BD}" +EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "MicroPlumberd.Migration.Runner", "MicroPlumberd.Migration.Runner\MicroPlumberd.Migration.Runner.csproj", "{6820868E-EF34-4CE6-9F64-0544A9FD9ECB}" +EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "MicroPlumberd.Migration.Tests", "MicroPlumberd.Migration.Tests\MicroPlumberd.Migration.Tests.csproj", "{9E26B841-4CB8-44A7-8847-4E3B504B52F4}" +EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "MicroPlumberd.Migration.Scripting", "MicroPlumberd.Migration.Scripting\MicroPlumberd.Migration.Scripting.csproj", "{26E606D5-2DD5-42AB-8426-64F97C6BA194}" +EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "MicroPlumberd.Rewrite.Tests", "MicroPlumberd.Rewrite.Tests\MicroPlumberd.Rewrite.Tests.csproj", "{15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}" +EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "MicroPlumberd.Rewrite", "MicroPlumberd.Rewrite\MicroPlumberd.Rewrite.csproj", "{11B13FE4-AF80-4801-9A4E-58752001A8D2}" +EndProject Global GlobalSection(SolutionConfigurationPlatforms) = preSolution Debug|Any CPU = Debug|Any CPU @@ -443,6 +455,78 @@ Global {0F09FC14-E906-4FF0-812B-25DF6B2C0D36}.Release|x64.Build.0 = Release|Any CPU {0F09FC14-E906-4FF0-812B-25DF6B2C0D36}.Release|x86.ActiveCfg = Release|Any CPU {0F09FC14-E906-4FF0-812B-25DF6B2C0D36}.Release|x86.Build.0 = Release|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Debug|Any CPU.Build.0 = Debug|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Debug|x64.ActiveCfg = Debug|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Debug|x64.Build.0 = Debug|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Debug|x86.ActiveCfg = Debug|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Debug|x86.Build.0 = Debug|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Release|Any CPU.ActiveCfg = Release|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Release|Any CPU.Build.0 = Release|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Release|x64.ActiveCfg = Release|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Release|x64.Build.0 = Release|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Release|x86.ActiveCfg = Release|Any CPU + {83CEAEE1-EE88-4D5C-B938-97D8123C51BD}.Release|x86.Build.0 = Release|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Debug|Any CPU.Build.0 = Debug|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Debug|x64.ActiveCfg = Debug|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Debug|x64.Build.0 = Debug|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Debug|x86.ActiveCfg = Debug|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Debug|x86.Build.0 = Debug|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Release|Any CPU.ActiveCfg = Release|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Release|Any CPU.Build.0 = Release|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Release|x64.ActiveCfg = Release|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Release|x64.Build.0 = Release|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Release|x86.ActiveCfg = Release|Any CPU + {6820868E-EF34-4CE6-9F64-0544A9FD9ECB}.Release|x86.Build.0 = Release|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Debug|Any CPU.Build.0 = Debug|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Debug|x64.ActiveCfg = Debug|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Debug|x64.Build.0 = Debug|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Debug|x86.ActiveCfg = Debug|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Debug|x86.Build.0 = Debug|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Release|Any CPU.ActiveCfg = Release|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Release|Any CPU.Build.0 = Release|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Release|x64.ActiveCfg = Release|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Release|x64.Build.0 = Release|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Release|x86.ActiveCfg = Release|Any CPU + {9E26B841-4CB8-44A7-8847-4E3B504B52F4}.Release|x86.Build.0 = Release|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Debug|Any CPU.Build.0 = Debug|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Debug|x64.ActiveCfg = Debug|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Debug|x64.Build.0 = Debug|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Debug|x86.ActiveCfg = Debug|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Debug|x86.Build.0 = Debug|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Release|Any CPU.ActiveCfg = Release|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Release|Any CPU.Build.0 = Release|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Release|x64.ActiveCfg = Release|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Release|x64.Build.0 = Release|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Release|x86.ActiveCfg = Release|Any CPU + {26E606D5-2DD5-42AB-8426-64F97C6BA194}.Release|x86.Build.0 = Release|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Debug|Any CPU.Build.0 = Debug|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Debug|x64.ActiveCfg = Debug|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Debug|x64.Build.0 = Debug|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Debug|x86.ActiveCfg = Debug|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Debug|x86.Build.0 = Debug|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Release|Any CPU.ActiveCfg = Release|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Release|Any CPU.Build.0 = Release|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Release|x64.ActiveCfg = Release|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Release|x64.Build.0 = Release|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Release|x86.ActiveCfg = Release|Any CPU + {15DECE8B-BCA9-4C02-B8A9-83F82FB86CDF}.Release|x86.Build.0 = Release|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Debug|Any CPU.Build.0 = Debug|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Debug|x64.ActiveCfg = Debug|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Debug|x64.Build.0 = Debug|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Debug|x86.ActiveCfg = Debug|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Debug|x86.Build.0 = Debug|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Release|Any CPU.ActiveCfg = Release|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Release|Any CPU.Build.0 = Release|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Release|x64.ActiveCfg = Release|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Release|x64.Build.0 = Release|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Release|x86.ActiveCfg = Release|Any CPU + {11B13FE4-AF80-4801-9A4E-58752001A8D2}.Release|x86.Build.0 = Release|Any CPU EndGlobalSection GlobalSection(SolutionProperties) = preSolution HideSolutionNode = FALSE diff --git a/src/MicroPlumberd/ISubscriptionState.cs b/src/MicroPlumberd/ISubscriptionState.cs new file mode 100644 index 0000000..c9ada19 --- /dev/null +++ b/src/MicroPlumberd/ISubscriptionState.cs @@ -0,0 +1,79 @@ +using KurrentDB.Client; + +namespace MicroPlumberd; + +/// +/// The seam that lets ONE subscription loop (SubscriptionRunner.WithHandler) serve both merge sources — +/// a projection-backed output stream (StreamSubscriptionState, via SubscribeToStream) and a +/// KurrentDB user-defined index (, via filtered SubscribeToAll). The +/// only per-source differences are which subscribe call is made and how the resume position is computed from a +/// delivered ; both are hidden here so the loop body (dispatch, CaughtUp, +/// resubscribe-with-backoff) is literally reused, not forked. See +/// docs/design-index-backed-merge-streams.md — "Why the loop is shared, not forked". +/// +interface ISubscriptionState : IDisposable +{ + /// Human-readable name of the subscribed source (for logs / operation context). + string StreamName { get; } + + /// Cancellation token that stops the subscription loop. + CancellationToken CancellationToken { get; } + + /// Opens the subscription from the current resume position (SubscribeToStream OR filtered SubscribeToAll). + KurrentDBClient.StreamSubscriptionResult Subscribe(); + + /// Advances the in-memory resume position after an event was successfully dispatched. + void Advance(ResolvedEvent e); +} + +/// +/// Index-backed merge subscription state (catch-up, client-checkpointed). Subscribes to filtered $all over +/// StreamFilter.Prefix($idx-user-<name>) with resolveLinkTos:true, resumes from the last +/// delivered link's $all (FromAll.After(pos)) — the ONLY working mechanism +/// (SPIKE-2b/SPIKE-3/SPIKE-9: a direct SubscribeToStream on the index name silently delivers nothing). +/// +sealed class IndexSubscriptionState : ISubscriptionState +{ + private readonly KurrentDBClient _client; + private readonly SubscriptionFilterOptions _filter; + private readonly UserCredentials? _userCredentials; + private IDisposable? _subscription; + private FromAll _position; + + /// The gRPC client used for the filtered $all subscription. + /// The index read stream — $idx-user-<name> — used as the prefix filter. + /// Boot start position (FromAll.Start rebuilds; FromAll.End tails only). + /// Optional per-subscription credentials. + /// Stops the subscription loop. + public IndexSubscriptionState(KurrentDBClient client, string indexStream, FromAll start, + UserCredentials? userCredentials, CancellationToken cancellationToken) + { + _client = client ?? throw new ArgumentNullException(nameof(client)); + StreamName = indexStream ?? throw new ArgumentNullException(nameof(indexStream)); + _filter = new SubscriptionFilterOptions(StreamFilter.Prefix(indexStream)); + _position = start; + _userCredentials = userCredentials; + CancellationToken = cancellationToken; + } + + public string StreamName { get; } + + public CancellationToken CancellationToken { get; } + + public KurrentDBClient.StreamSubscriptionResult Subscribe() + { + var result = _client.SubscribeToAll(_position, resolveLinkTos: true, filterOptions: _filter, + userCredentials: _userCredentials, cancellationToken: CancellationToken); + _subscription = result; + return result; + } + + public void Advance(ResolvedEvent e) + { + // The resume key is the link's $all Position — NOT an index-stream revision (SPIKE-3). Resolved reads may + // carry a null OriginalPosition for some system frames; keep the last known position in that case. + if (e.OriginalPosition is { } p) _position = FromAll.After(p); + } + + public void Dispose() => _subscription?.Dispose(); +} diff --git a/src/MicroPlumberd/KurrentHttpEndpoint.cs b/src/MicroPlumberd/KurrentHttpEndpoint.cs new file mode 100644 index 0000000..47c4907 --- /dev/null +++ b/src/MicroPlumberd/KurrentHttpEndpoint.cs @@ -0,0 +1,89 @@ +using System.Net.Http.Headers; +using System.Text; +using System.Text.RegularExpressions; +using KurrentDB.Client; + +namespace MicroPlumberd; + +/// +/// Derives the HTTP(S) base address + basic-auth credentials of a KurrentDB node from its gRPC/esdb +/// connection string (or from an already-parsed ), and builds an +/// for the management APIs (user-defined indexes over /v2/indexes) the +/// gRPC client does not expose. +/// +/// +/// Relocated into core (from MicroPlumberd.Migration) so the LIVE index-backed subscription path and the +/// offline migration path share ONE endpoint-parsing implementation (see +/// docs/design-index-backed-merge-streams.md — "Assembly layering"). The offline path parses a connection +/// string via ; the live path derives the same triple from the running engine's +/// via . +/// +public static class KurrentHttpEndpoint +{ + /// Parses the first host + credentials + TLS flag out of a KurrentDB connection string. + public static (Uri BaseUri, string User, string Pass) Parse(string connectionString) + { + var m = Regex.Match(connectionString, + @"^(?[a-zA-Z0-9+]+)://(?:(?[^:@/]+):(?[^@/]+)@)?(?[^/?]+)(?:/[^?]*)?(?:\?(?.*))?$"); + if (!m.Success) + throw new ArgumentException($"Unrecognised KurrentDB connection string: '{connectionString}'."); + + var user = m.Groups["user"].Success ? m.Groups["user"].Value : "admin"; + var pass = m.Groups["pass"].Success ? m.Groups["pass"].Value : "changeit"; + var firstHost = m.Groups["hosts"].Value.Split(',', StringSplitOptions.RemoveEmptyEntries)[0].Trim(); + if (!firstHost.Contains(':')) firstHost += ":2113"; + + var tls = true; + if (m.Groups["query"].Success) + foreach (var kv in m.Groups["query"].Value.Split('&', StringSplitOptions.RemoveEmptyEntries)) + { + var parts = kv.Split('=', 2); + if (parts.Length == 2 && parts[0].Equals("tls", StringComparison.OrdinalIgnoreCase)) + tls = !parts[1].Equals("false", StringComparison.OrdinalIgnoreCase); + } + + return (new Uri($"{(tls ? "https" : "http")}://{firstHost}/"), user, pass); + } + + /// + /// Derives the HTTP management base address + basic-auth credentials from an already-configured + /// (the shape the running holds). Uses the + /// same node address the gRPC client connects to (ConnectivitySettings.Address, or the first gossip + /// seed when no explicit address is set) so the index HTTP calls and the gRPC reads target the same node. + /// + public static (Uri BaseUri, string User, string Pass) FromSettings(KurrentDBClientSettings settings) + { + ArgumentNullException.ThrowIfNull(settings); + var cs = settings.ConnectivitySettings; + var address = cs?.Address; + if (address is null) + { + var seed = cs?.GossipSeeds is { Length: > 0 } seeds ? seeds[0] : null; + if (seed is null) + throw new InvalidOperationException( + "Cannot derive a KurrentDB HTTP endpoint: the client settings have neither ConnectivitySettings.Address " + + "nor a gossip seed. Index-backed merge needs the node's HTTP address to manage /v2/indexes."); + var insecure = cs?.Insecure ?? false; + address = new Uri($"{(insecure ? "http" : "https")}://{seed}/"); + } + + // Keep only scheme://host:port/ — the /v2/indexes URIs are resolved relative to this base. + var baseUri = new Uri(address.GetLeftPart(UriPartial.Authority) + "/"); + var user = settings.DefaultCredentials?.Username ?? "admin"; + var pass = settings.DefaultCredentials?.Password ?? "changeit"; + return (baseUri, user, pass); + } + + /// Builds an with basic auth that accepts the node's dev certificate. + public static HttpClient CreateClient(string user, string pass) + { + var handler = new HttpClientHandler + { + ServerCertificateCustomValidationCallback = HttpClientHandler.DangerousAcceptAnyServerCertificateValidator + }; + var http = new HttpClient(handler); + var basic = Convert.ToBase64String(Encoding.ASCII.GetBytes($"{user}:{pass}")); + http.DefaultRequestHeaders.Authorization = new AuthenticationHeaderValue("Basic", basic); + return http; + } +} diff --git a/src/MicroPlumberd/PlumberEngine.cs b/src/MicroPlumberd/PlumberEngine.cs index 41435fa..24e239a 100644 --- a/src/MicroPlumberd/PlumberEngine.cs +++ b/src/MicroPlumberd/PlumberEngine.cs @@ -40,10 +40,14 @@ public class PlumberEngine : IPlumberReadOnlyConfig private readonly VersionDuckTyping _versionTyping = new(); private readonly Func> _errorHandle; private ProjectionRegister? _projectionRegister; + private readonly KurrentDBClientSettings _settings; + private UserDefinedIndex? _userDefinedIndex; + private IIndexDefinitionReconciler? _indexReconciler; internal PlumberEngine(KurrentDBClientSettings settings, PlumberConfig? config = null) { config ??= new PlumberConfig(); + _settings = settings; Client = new KurrentDBClient(settings); PersistentSubscriptionClient = new KurrentDBPersistentSubscriptionsClient(settings); ProjectionManagementClient = new KurrentDBProjectionManagementClient(settings); @@ -297,6 +301,80 @@ public async Task SubscribeEventHandler(TypeEve return sub; } + /// The relocated user-defined-index lifecycle primitive, built from this engine's connection settings. + internal UserDefinedIndex UserDefinedIndex => + _userDefinedIndex ??= BuildUserDefinedIndex(); + + private UserDefinedIndex BuildUserDefinedIndex() + { + var (baseUri, user, pass) = KurrentHttpEndpoint.FromSettings(_settings); + return new UserDefinedIndex(baseUri, user, pass, ServiceProvider?.GetService>()); + } + + private IIndexDefinitionReconciler IndexReconciler => + _indexReconciler ??= new UserDefinedIndexReconciler(UserDefinedIndex, + ServiceProvider?.GetService>()); + + /// + /// INDEX-BACKED sibling of : + /// sources the merged stream from a KurrentDB user-defined index (filtered $all) instead of a + /// fromStreams(...).linkTo(outputStream) join projection. Catch-up only (persistent subscriptions + /// cannot see index links — SPIKE-7). Same handler dispatch, ICaughtUpHandler and error-backoff as the + /// projection path — only the event SOURCE differs. Opt-in; the projection path stays the default. + /// + /// Weaker CaughtUp semantics than the projection path. The history→live boundary tracks the $all + /// position, so while the index is still backfilling, ICaughtUpHandler.CaughtUp() can fire BEFORE all + /// historical links are delivered (they arrive as "live"). No-loss and commit ordering still hold; a read model + /// that treats CaughtUp as a "fully caught up / authoritative" signal should stay projection-backed. + /// + /// + /// The catch-up read model to index-back. + /// Optional handler instance; resolved from DI when null. + /// Optional output-stream name (convention-derived when null) — the index-name seed. + /// Only Start (default, rebuild) or End (tail-only) are valid on filtered $all. + /// Cancellation token. + /// A disposable subscription that stops the index-backed tail when disposed. + public async Task SubscribeEventHandlerViaIndex(TEventHandler? eh = null, + string? outputStream = null, FromRelativeStreamPosition? start = null, CancellationToken token = default) + where TEventHandler : class, IEventHandler, ITypeRegister + { + outputStream ??= Conventions.OutputStreamModelConvention(typeof(TEventHandler)); + var eventTypes = _typeHandlerRegisters.GetEventNamesFor().ToHashSet(StringComparer.Ordinal); + var tailOnly = IsTailOnlyStart(start); + var fromAll = tailOnly ? FromAll.End : FromAll.Start; + + var name = await IndexReconciler.ReconcileAsync(outputStream, eventTypes, token).ConfigureAwait(false); + var indexStream = UserDefinedIndex.IndexStream(name); + + var logger = ServiceProvider?.GetService>(); + logger?.LogInformation( + "Index-backed subscription: handler {Handler} → index '{Index}' → stream {IndexStream} → types [{Types}] (start {Start}).", + typeof(TEventHandler).Name, name, indexStream, string.Join(",", eventTypes), + tailOnly ? "End" : "Start"); + + var state = new IndexSubscriptionState(Client, indexStream, fromAll, _settings.DefaultCredentials, token); + var runner = new SubscriptionRunner(this, state); + var mapFunc = _typeHandlerRegisters.GetEventNameConverterFor()!; + if (eh == null) + await runner.WithHandler(mapFunc); + else + await runner.WithHandler(eh, mapFunc); + return runner; + } + + // Only Start/End are meaningful on a filtered-$all subscription; a specific-revision start is rejected loud + // (it has no meaning on the index path). FromStream.Start → false (rebuild), FromStream.End → true (tail-only). + private static bool IsTailOnlyStart(FromRelativeStreamPosition? start) + { + if (start is null) return false; + var s = start.Value; + if (s.StartPosition == FromStream.Start) return false; + if (s.StartPosition == FromStream.End) return true; + throw new InvalidOperationException( + "Index-backed merge supports only Start or End as a start position; a specific stream revision has no " + + "meaning on a filtered-$all subscription. See docs/design-index-backed-merge-streams.md."); + } + /// /// Subscribes an event handler persistently with at-least-once delivery semantics. /// diff --git a/src/MicroPlumberd/SubscriptionRunner.cs b/src/MicroPlumberd/SubscriptionRunner.cs index 8020202..c74b386 100644 --- a/src/MicroPlumberd/SubscriptionRunner.cs +++ b/src/MicroPlumberd/SubscriptionRunner.cs @@ -6,9 +6,12 @@ namespace MicroPlumberd; /// -/// Represents the state of a subscription runner. +/// Stream-backed (projection output-stream) subscription state — the existing behaviour, refactored behind +/// so the shared loop can also drive the index-backed +/// . Subscribes via SubscribeToStream and resumes on the stream-local +/// revision (FromStream.After(OriginalEventNumber)). /// -record SubscriptionRunnerState : IDisposable +record SubscriptionRunnerState : ISubscriptionState { private IDisposable? _subscription; @@ -24,12 +27,23 @@ public SubscriptionRunnerState(FromStream initialPosition, KurrentDBClient clien public KurrentDBClient.StreamSubscriptionResult Subscribe() { + // Loud guard against the SPIKE-2a/SPIKE-9 footgun: an index read stream is NOT directly subscribable — + // SubscribeToStream("$idx-user-…") silently delivers ZERO events. Fail LOUD instead of hanging silent. + // Index streams are ONLY ever read via filtered $all (IndexSubscriptionState). + if (StreamName.StartsWith(UserDefinedIndex.IndexStreamPrefixRoot, StringComparison.Ordinal)) + throw new InvalidOperationException( + $"'{StreamName}' is a user-defined-index read stream and cannot be consumed via SubscribeToStream " + + "(it silently delivers zero events — SPIKE-2a/SPIKE-9). Use IndexSubscriptionState (filtered " + + "SubscribeToAll + StreamFilter.Prefix) instead. See docs/design-index-backed-merge-streams.md."); + var result = _client.SubscribeToStream(StreamName, Position, true, UserCredentials, CancellationToken); _subscription = result; return result; } + + public void Advance(ResolvedEvent e) => Position = FromStream.After(e.OriginalEventNumber); + public FromStream Position { get; set; } - public IEventHandler Handler { get; set; } private readonly FromStream _initialPosition; private readonly KurrentDBClient _client; public string StreamName { get; init; } @@ -41,7 +55,7 @@ public void Dispose() _subscription?.Dispose(); } - + }; class SubscriptionSeeker(PlumberEngine plumber, string streamName, FromRelativeStreamPosition start, UserCredentials? userCredentials = null, CancellationToken cancellationToken = default) : ISubscriptionRunner @@ -143,7 +157,7 @@ public FailFastException(string msg, Exception inner) : base(msg,inner) } } -class SubscriptionRunner(PlumberEngine plumber, SubscriptionRunnerState subscription) : ISubscriptionRunner +class SubscriptionRunner(PlumberEngine plumber, ISubscriptionState subscription) : ISubscriptionRunner { public async Task WithHandler(T model) where T : IEventHandler, ITypeRegister @@ -159,7 +173,6 @@ public async Task WithHandler(T model, TypeEventConverter func) public async Task WithHandler(IEventHandler model, TypeEventConverter func) { - subscription.Handler = model; await Task.Factory.StartNew(async (_) => { var l = plumber.Config.ServiceProvider.GetService>(); @@ -174,7 +187,7 @@ await Task.Factory.StartNew(async (_) => { case StreamMessage.Event(var e): await OnEvent(func, e, model); - subscription.Position = FromStream.After(e.OriginalEventNumber); + subscription.Advance(e); break; case StreamMessage.CaughtUp: l?.LogDebug($"Subscription '{subscription.StreamName}' caught up."); diff --git a/src/MicroPlumberd/UserDefinedIndex.cs b/src/MicroPlumberd/UserDefinedIndex.cs new file mode 100644 index 0000000..b468f0e --- /dev/null +++ b/src/MicroPlumberd/UserDefinedIndex.cs @@ -0,0 +1,338 @@ +using System.Collections.Concurrent; +using System.Net; +using System.Security.Cryptography; +using System.Text; +using System.Text.Json.Nodes; +using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MicroPlumberd; + +/// +/// The relocated, single implementation of the KurrentDB 26.1 USER-DEFINED INDEX lifecycle primitives shared by +/// the LIVE index-backed merge subscription path (core) and the offline migration reader +/// (MicroPlumberd.Migration.UserDefinedIndexSource, which delegates here). Creates/ensures an index over a +/// set of event types (), deletes a superseded one (), reads a +/// stored filter () and lists index names (), plus the +/// pure name/filter/stream helpers (, , +/// , ). +/// +/// +/// Read the index via filtered $all only. An index's links live in $all; the +/// $idx-user-{name} name is NOT a directly subscribable/readable stream (SPIKE-2a/SPIKE-9). Consume it via +/// SubscribeToAll/ReadAllAsync + — +/// does exactly this, and StreamSubscriptionState guards against the +/// silent-zero footgun. +/// Creation is idempotent. POSTs /v2/indexes/{name}; an existing +/// index (HTTP 409 INDEX_ALREADY_EXISTS, or an identical 200) is a no-op. A DIFFERENT filter on an +/// existing name is left as-is with a logged warning — KurrentDB does not redefine an index in place (SPIKE-5: +/// re-POST → 409), so a filter change means a NEW name (see + +/// UserDefinedIndexReconciler). +/// Every seam logs (index name + the failing URI on error) via the injected , +/// defaulting to so the type is safe to construct anywhere (incl. WASM: no +/// reload-on-change, no ambient statics). +/// +public sealed class UserDefinedIndex +{ + /// The fixed prefix of an index read stream: $idx-user-. + public const string IndexStreamPrefixRoot = "$idx-user-"; + + /// The prefix of a reconciler-managed index NAME: mpidx- (before the normalized output + hash). + public const string ManagedIndexNamePrefix = "mpidx-"; + + private readonly Uri _httpBase; + private readonly string _user; + private readonly string _pass; + private readonly ILogger _logger; + + /// + /// + /// The node HTTP(S) management base — see . + /// Basic-auth user for the management API. + /// Basic-auth password for the management API. + /// Optional; defaults to . + public UserDefinedIndex(Uri httpBase, string user, string pass, ILogger? logger = null) + { + _httpBase = httpBase ?? throw new ArgumentNullException(nameof(httpBase)); + _user = user ?? throw new ArgumentNullException(nameof(user)); + _pass = pass ?? throw new ArgumentNullException(nameof(pass)); + _logger = logger ?? NullLogger.Instance; + } + + /// The read stream ($idx-user-{name}) that carries 's indexed links. + public static string IndexStream(string name) => IndexStreamPrefixRoot + name; + + /// + /// Creates the index if it does not already exist (idempotent). Returns without waiting for the backfill — + /// the LIVE path subscribes-then-tails immediately (no count gate); the offline path calls its own + /// count-convergence readiness gate. No-op when is true. + /// + public async Task EnsureAsync(string name, IReadOnlySet eventTypes, bool dryRun = false, + CancellationToken ct = default) + { + ArgumentNullException.ThrowIfNull(eventTypes); + name = NormalizeName(name); + var filter = BuildFilter(eventTypes); + + if (dryRun) + { + _logger.LogInformation("DRY RUN: would create user-defined index '{Name}' with filter [{Filter}].", + name, filter); + return; + } + + var url = new Uri(_httpBase, $"v2/indexes/{Uri.EscapeDataString(name)}"); + using var http = KurrentHttpEndpoint.CreateClient(_user, _pass); + var payload = new JsonObject { ["filter"] = filter, ["start"] = true }; + using var content = new StringContent(payload.ToJsonString(), Encoding.UTF8, "application/json"); + + HttpResponseMessage resp; + try { resp = await http.PostAsync(url, content, ct).ConfigureAwait(false); } + catch (Exception ex) + { + _logger.LogError(ex, "Failed to POST user-defined index '{Name}' at {Uri}.", name, url); + throw; + } + + if (resp.IsSuccessStatusCode) + { + _logger.LogInformation("Created (or confirmed) user-defined index '{Name}' with filter [{Filter}].", + name, filter); + return; + } + + var body = await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + if (resp.StatusCode == HttpStatusCode.Conflict) // 409 INDEX_ALREADY_EXISTS — idempotent success + { + await WarnOnFilterDriftAsync(http, name, filter, ct).ConfigureAwait(false); + _logger.LogInformation("User-defined index '{Name}' already exists — reusing it.", name); + return; + } + + _logger.LogError("Creating user-defined index '{Name}' at {Uri} failed: HTTP {Code} {Body}.", + name, url, (int)resp.StatusCode, body); + throw new InvalidOperationException( + $"Creating user-defined index '{name}' failed: HTTP {(int)resp.StatusCode} at {url} — {body}"); + } + + /// + /// Deletes an index by name (idempotent — a missing index is treated as already-deleted). Returns + /// true when the server confirmed the delete (HTTP 2xx) or the index was already gone (404). + /// Used by the reconciler to retire a superseded (old-hash) index after a definition change. + /// + public async Task DeleteAsync(string name, CancellationToken ct = default) + { + name = NormalizeName(name); + var url = new Uri(_httpBase, $"v2/indexes/{Uri.EscapeDataString(name)}"); + using var http = KurrentHttpEndpoint.CreateClient(_user, _pass); + + HttpResponseMessage resp; + try { resp = await http.DeleteAsync(url, ct).ConfigureAwait(false); } + catch (Exception ex) + { + _logger.LogWarning(ex, "Failed to DELETE user-defined index '{Name}' at {Uri}.", name, url); + return false; + } + + if (resp.IsSuccessStatusCode || resp.StatusCode == HttpStatusCode.NotFound) + { + _logger.LogInformation("Deleted user-defined index '{Name}' (HTTP {Code}).", name, (int)resp.StatusCode); + return true; + } + + var body = await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + _logger.LogWarning("Deleting user-defined index '{Name}' at {Uri} failed: HTTP {Code} {Body}.", + name, url, (int)resp.StatusCode, body); + return false; + } + + /// Reads the stored filter for an index, or null when the index does not exist / has no filter. + public async Task GetFilterAsync(string name, CancellationToken ct = default) + { + name = NormalizeName(name); + var url = new Uri(_httpBase, $"v2/indexes/{Uri.EscapeDataString(name)}"); + using var http = KurrentHttpEndpoint.CreateClient(_user, _pass); + try + { + using var resp = await http.GetAsync(url, ct).ConfigureAwait(false); + if (!resp.IsSuccessStatusCode) return null; + var json = await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + return JsonNode.Parse(json)?["index"]?["filter"]?.GetValue(); + } + catch (Exception ex) + { + _logger.LogWarning(ex, "Could not read filter for user-defined index '{Name}' at {Uri}.", name, url); + return null; + } + } + + /// + /// Lists the names of all user-defined indexes on the node (best-effort — an unparseable or missing list + /// endpoint logs a warning and yields nothing, since orphan cleanup is best-effort per design). Used by the + /// reconciler to find superseded mpidx-<output>-<hash> indexes to delete. + /// + public async Task> ListNamesAsync(CancellationToken ct = default) + { + var url = new Uri(_httpBase, "v2/indexes"); + using var http = KurrentHttpEndpoint.CreateClient(_user, _pass); + try + { + using var resp = await http.GetAsync(url, ct).ConfigureAwait(false); + if (!resp.IsSuccessStatusCode) + { + _logger.LogWarning("Listing user-defined indexes at {Uri} returned HTTP {Code}; orphan cleanup skipped.", + url, (int)resp.StatusCode); + return Array.Empty(); + } + var json = await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + return ParseIndexNames(JsonNode.Parse(json)); + } + catch (Exception ex) + { + _logger.LogWarning(ex, "Could not list user-defined indexes at {Uri}; orphan cleanup skipped.", url); + return Array.Empty(); + } + } + + // Tolerant parse: the list endpoint may return {"indexes":[{"name":..}]}, {"indexes":["name",..]} or a bare + // array. Collect any "name" fields and/or bare string entries under a top-level array-ish node. + private static IReadOnlyList ParseIndexNames(JsonNode? root) + { + var names = new List(); + JsonArray? arr = root as JsonArray + ?? root?["indexes"] as JsonArray + ?? root?["index"] as JsonArray; + if (arr is null) return names; + foreach (var item in arr) + { + switch (item) + { + case JsonValue v when v.TryGetValue(out var s) && !string.IsNullOrEmpty(s): + names.Add(s); + break; + case JsonObject o when o["name"]?.GetValue() is { Length: > 0 } n: + names.Add(n); + break; + } + } + return names; + } + + // ---- pure helpers -------------------------------------------------------------------------------------- + + /// + /// The reconciler-managed index NAME for a merge: mpidx-<normalized-output>-<hash8-of-filter>. + /// hash8 is the first 8 lower-hex chars of SHA-256(filter) (the filter is ordinal-sorted, so it is + /// STABLE for a given event-type SET). Hashing the FILTER (not the output-stream name) is what makes a changed + /// event-type set deterministically produce a NEW name — the trigger for create-new-and-swap. + /// + public static string IndexNameFor(string outputStream, string? filter) + { + var norm = NormalizeName(outputStream); + var bytes = SHA256.HashData(Encoding.UTF8.GetBytes(filter ?? string.Empty)); + var hash8 = Convert.ToHexString(bytes, 0, 4).ToLowerInvariant(); // 4 bytes -> 8 hex chars + return $"{ManagedIndexNamePrefix}{norm}-{hash8}"; + } + + /// The mpidx-<normalized-output>- prefix shared by every managed index for one output stream. + public static string ManagedNamePrefixFor(string outputStream) => + $"{ManagedIndexNamePrefix}{NormalizeName(outputStream)}-"; + + // Process-wide owner of each managed-index BASE (the normalized output stream). Guards the data-destructive + // punctuation-collision edge below. + private static readonly ConcurrentDictionary _managedBaseOwners = new(StringComparer.Ordinal); + + /// + /// Claims the managed-index BASE (the normalized output stream) for in this + /// process, throwing if a DIFFERENT output stream already owns the same base. A punctuation collision — e.g. + /// "Foo.1" and "Foo-1" both normalize to "foo-1" — would make two read models share the + /// mpidx-<base>-* prefix, so orphan-cleanup for one could DELETE the other's live index. This makes + /// that fail LOUD at reconcile instead of silently deleting the wrong index. Idempotent for the SAME output + /// stream (re-reconciling one read model never collides with itself). + /// + public static void RegisterManagedBase(string outputStream) + { + if (string.IsNullOrWhiteSpace(outputStream)) + throw new ArgumentException("Output stream must be non-empty.", nameof(outputStream)); + var baseName = NormalizeName(outputStream); + var owner = _managedBaseOwners.GetOrAdd(baseName, outputStream); + if (!string.Equals(owner, outputStream, StringComparison.Ordinal)) + throw new InvalidOperationException( + $"Managed user-defined-index name collision: output streams '{owner}' and '{outputStream}' both " + + $"normalize to '{baseName}', so their managed indexes ('{ManagedIndexNamePrefix}{baseName}-*') " + + "would collide and orphan-cleanup could DELETE the wrong read model's index. Rename one output " + + "stream so they normalize distinctly."); + } + + /// + /// Builds the index filter as the single-argument JavaScript arrow function KurrentDB requires: + /// rec => rec.schema.name == "A" || rec.schema.name == "B". An empty set ⇒ null (index all records). + /// Ordinal-sorted so the filter (and thus its hash) is STABLE for a given set; each name is JS-escaped so a + /// quote/backslash/control char cannot break out of the literal. + /// + public static string? BuildFilter(IReadOnlySet eventTypes) + { + ArgumentNullException.ThrowIfNull(eventTypes); + if (eventTypes.Count == 0) return null; + var terms = eventTypes.OrderBy(t => t, StringComparer.Ordinal) + .Select(t => $"rec.schema.name == \"{EscapeJsString(t)}\""); + return "rec => " + string.Join(" || ", terms); + } + + /// + /// Normalises an arbitrary name (e.g. an [OutputStream] id like FooModel_v1) to a valid KurrentDB + /// index name: lower-case; every character outside [a-z0-9_] becomes -. Idempotent on names it + /// already produced (so re-normalising a mpidx-… name is a no-op). + /// + public static string NormalizeName(string raw) + { + if (string.IsNullOrWhiteSpace(raw)) throw new ArgumentException("Index name must be non-empty.", nameof(raw)); + var sb = new StringBuilder(raw.Length); + foreach (var ch in raw) + { + var c = char.ToLowerInvariant(ch); + sb.Append(c is (>= 'a' and <= 'z') or (>= '0' and <= '9') or '_' or '-' ? c : '-'); + } + return sb.ToString(); + } + + /// Escapes a string for a JavaScript double-quoted literal (backslash, quote, control chars). + internal static string EscapeJsString(string s) + { + var sb = new StringBuilder(s.Length + 8); + foreach (var ch in s) + sb.Append(ch switch + { + '\\' => "\\\\", + '"' => "\\\"", + '\n' => "\\n", + '\r' => "\\r", + '\t' => "\\t", + < ' ' => $"\\u{(int)ch:x4}", + _ => ch.ToString() + }); + return sb.ToString(); + } + + private async Task WarnOnFilterDriftAsync(HttpClient http, string name, string? requestedFilter, CancellationToken ct) + { + var url = new Uri(_httpBase, $"v2/indexes/{Uri.EscapeDataString(name)}"); + try + { + using var resp = await http.GetAsync(url, ct).ConfigureAwait(false); + if (!resp.IsSuccessStatusCode) return; + var json = await resp.Content.ReadAsStringAsync(ct).ConfigureAwait(false); + var existing = JsonNode.Parse(json)?["index"]?["filter"]?.GetValue(); + var existingNorm = string.IsNullOrEmpty(existing) ? null : existing; + if (!string.Equals(existingNorm, requestedFilter, StringComparison.Ordinal)) + _logger.LogWarning( + "User-defined index '{Name}' already exists with a DIFFERENT filter (existing [{Existing}], " + + "requested [{Requested}]) — the existing definition is kept.", name, existingNorm ?? "", + requestedFilter ?? ""); + } + catch (Exception ex) + { + _logger.LogWarning(ex, "Could not read existing filter for user-defined index '{Name}'.", name); + } + } +} diff --git a/src/MicroPlumberd/UserDefinedIndexReconciler.cs b/src/MicroPlumberd/UserDefinedIndexReconciler.cs new file mode 100644 index 0000000..8c2a60b --- /dev/null +++ b/src/MicroPlumberd/UserDefinedIndexReconciler.cs @@ -0,0 +1,59 @@ +using Microsoft.Extensions.Logging; + +namespace MicroPlumberd; + +/// +/// Idempotently "ensure the current merge index exists and return the index NAME to subscribe to". Kept behind a +/// seam so a future in-place-redefine strategy could drop in IF/when KurrentDB ships index-definition updates — +/// today that is impossible (SPIKE-5: a redefine POST is 409-rejected), so +/// (create-new-and-swap) is the sole implementation. +/// +interface IIndexDefinitionReconciler +{ + /// + /// Ensures the current filter-hashed index for / + /// exists (creating a new one on an event-type-set change, reusing an unchanged one), deletes superseded + /// managed indexes for the same output stream, and returns the index NAME the caller should subscribe to. + /// + Task ReconcileAsync(string outputStream, IReadOnlySet eventTypes, CancellationToken ct); +} + +/// +/// THE settled, permanent lifecycle for an index-backed merge (SPIKE-5): a filter-hashed index name +/// (mpidx-<output>-<hash8>) means a changed event-type set deterministically yields a NEW name, +/// so the read model rebuilds from the new merged view (the projection disable→update→enable analog). +/// Unchanged type set ⇒ same name ⇒ 409 reuse ⇒ nothing created/deleted/rebuilt. After ensuring the current +/// index, superseded mpidx-<output>-* indexes are DELETEd (best-effort — correctness never depends on +/// cleanup running; an un-deleted orphan is harmless). +/// +sealed class UserDefinedIndexReconciler(UserDefinedIndex index, ILogger? logger = null) : IIndexDefinitionReconciler +{ + public async Task ReconcileAsync(string outputStream, IReadOnlySet eventTypes, CancellationToken ct) + { + // Fail LOUD on a punctuation collision before doing any (potentially destructive) DELETE cleanup: two + // distinct output streams that normalize to the same managed base would share the mpidx--* prefix. + UserDefinedIndex.RegisterManagedBase(outputStream); + + var filter = UserDefinedIndex.BuildFilter(eventTypes); + var name = UserDefinedIndex.IndexNameFor(outputStream, filter); + + await index.EnsureAsync(name, eventTypes, ct: ct).ConfigureAwait(false); + + // Best-effort orphan cleanup: DELETE any managed index for THIS output stream whose hash suffix differs + // from the current one (an old event-type set superseded by this reconcile). Single-instance boot-time + // cleanup is safe (this app is the only reader and has already cut over to `name`). + var prefix = UserDefinedIndex.ManagedNamePrefixFor(outputStream); + foreach (var existing in await index.ListNamesAsync(ct).ConfigureAwait(false)) + { + if (existing.StartsWith(prefix, StringComparison.Ordinal) && + !string.Equals(existing, name, StringComparison.Ordinal)) + { + logger?.LogInformation("Reconciler deleting superseded index '{Old}' (current is '{Current}').", + existing, name); + await index.DeleteAsync(existing, ct).ConfigureAwait(false); + } + } + + return name; + } +}