From a4f271bcc9143d2b5444e7a8a5f25c86aa3d7422 Mon Sep 17 00:00:00 2001 From: nemeregiftbrown-byte Date: Tue, 29 Sep 2026 13:15:43 +0000 Subject: [PATCH 1/4] feat(devops): harden production Docker images and enable Next.js standalone (#1335) Backend runner now drops to the non-root `node` user, adds a native /health HEALTHCHECK, runs `node` directly as PID 1 for graceful SIGTERM handling, and purges the npm cache. Copied artifacts are chowned during COPY so no extra image layer is created. Frontend gains a dedicated multi-stage Dockerfile built on Next.js `output: "standalone"` (with a new dependency-free /health route), so the runtime image ships only the traced bundle instead of the whole workspace node_modules. A repo-root .dockerignore keeps that build context lean. docker-compose now builds and runs the frontend, wires depends_on to the backend's health probe, and probes 127.0.0.1 (busybox wget resolves `localhost` to IPv6 first, which the IPv4-bound servers do not answer). --- .dockerignore | 36 +++++++++++++++++ backend/Dockerfile | 38 +++++++++++++----- docker-compose.yml | 33 ++++++++++++++- frontend/Dockerfile | 69 ++++++++++++++++++++++++++++++++ frontend/next.config.ts | 5 +++ frontend/src/app/health/route.ts | 8 ++++ 6 files changed, 178 insertions(+), 11 deletions(-) create mode 100644 .dockerignore create mode 100644 frontend/Dockerfile create mode 100644 frontend/src/app/health/route.ts diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 00000000..d93a7aea --- /dev/null +++ b/.dockerignore @@ -0,0 +1,36 @@ +# Dockerignore for builds whose context is the repository root (the frontend +# image, issue #1335). The backend image uses backend/.dockerignore instead. +# Keeping the context lean makes builds faster and avoids leaking local state. + +# Version control / editor / CI +.git +.gitignore +.gitattributes +.github +.husky +.vscode +.idea + +# Dependencies and build output — reinstalled/rebuilt inside the image +node_modules +**/node_modules +frontend/.next +frontend/out +frontend/coverage +backend/dist +backend/coverage +backend/src/generated + +# Tests and docs are not needed to build the app image +**/__tests__/** +**/*.test.* +**/*.spec.* +e2e +docs +contracts + +# Secrets / local state +.env +.env.* +*.log +npm-debug.log* diff --git a/backend/Dockerfile b/backend/Dockerfile index f8f8a1cf..11f424cb 100644 --- a/backend/Dockerfile +++ b/backend/Dockerfile @@ -1,3 +1,8 @@ +# ============================================================================ +# Stage 1 — builder +# Full Node toolchain (devDependencies incl. tsc + prisma CLI) compiles the +# TypeScript and generates the Prisma client. Discarded from the final image. +# ============================================================================ FROM node:20-alpine@sha256:fb4cd12c85ee03686f6af5362a0b0d56d50c58a04632e6c0fb8363f609372293 AS builder WORKDIR /app @@ -11,7 +16,7 @@ WORKDIR /app COPY package*.json ./ COPY prisma ./prisma -RUN npm install +RUN npm install --no-audit --no-fund COPY tsconfig.json ./ COPY prisma.config.ts ./ @@ -19,6 +24,10 @@ COPY src ./src RUN npm run build +# ============================================================================ +# Stage 2 — production runner +# Minimal Alpine base, production-only dependencies, non-root `node` user. +# ============================================================================ FROM node:20-alpine@sha256:fb4cd12c85ee03686f6af5362a0b0d56d50c58a04632e6c0fb8363f609372293 AS runner WORKDIR /app @@ -30,11 +39,10 @@ ENV NODE_ENV=production # the dependency install. --ignore-scripts skips the @prisma/client # postinstall (we already COPY the generated client from the builder # below), so the runner never needs the schema to install dependencies. -# The schema is then copied AFTER install so it is on disk for the -# `npx prisma db push` step the workflow runs against this container, -# while keeping the install layer cached whenever prisma/ is unchanged. +# The npm cache is purged in the same layer so it never ships in the image. COPY package*.json ./ -RUN npm install --omit=dev --ignore-scripts +RUN npm install --omit=dev --ignore-scripts --no-audit --no-fund \ + && npm cache clean --force # Copy the prisma schema into the runner. The boot-and-check-health step # in .github/workflows/ci.yml runs `npx prisma db push` against this @@ -42,11 +50,21 @@ RUN npm install --omit=dev --ignore-scripts # out and the 60-iteration /health poll times the job out. COPY prisma ./prisma -COPY --from=builder /app/dist ./dist -COPY --from=builder /app/src/generated ./dist/generated -COPY --from=builder /app/prisma ./prisma -COPY prisma.config.ts ./ +COPY --from=builder --chown=node:node /app/dist ./dist +COPY --from=builder --chown=node:node /app/src/generated ./dist/generated +COPY --chown=node:node prisma.config.ts ./ + +# Drop root privileges for the running process. The `node` user (uid 1000) +# ships with the node:20-alpine base image — no extra user needs creating. +USER node EXPOSE 3001 -CMD ["npm", "start"] \ No newline at end of file +# Native healthcheck so Docker Compose / Kubernetes can gate traffic on +# readiness instead of only checking that the process is running. +HEALTHCHECK --interval=10s --timeout=5s --start-period=15s --retries=5 \ + CMD wget --no-verbose --tries=1 --spider http://127.0.0.1:3001/health || exit 1 + +# Exec node directly instead of `npm start` so the server runs as PID 1 and +# receives SIGTERM/SIGINT for the graceful-shutdown handler in src/index.ts. +CMD ["node", "dist/index.js"] diff --git a/docker-compose.yml b/docker-compose.yml index 58b04516..6c2c6484 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -40,10 +40,41 @@ services: postgres: condition: service_healthy healthcheck: - test: ["CMD", "wget", "--no-verbose", "--tries=1", "--spider", "http://localhost:3001/health"] + # Probe 127.0.0.1 explicitly: busybox wget resolves `localhost` to IPv6 + # first, which the IPv4-bound server does not answer. + test: ["CMD", "wget", "--no-verbose", "--tries=1", "--spider", "http://127.0.0.1:3001/health"] interval: 10s timeout: 5s retries: 5 + # Prisma + DB warm-up happens after the port opens; give it room before + # the first probe is counted against the retry budget. + start_period: 15s + restart: unless-stopped + + frontend: + build: + context: . + dockerfile: frontend/Dockerfile + args: + # NEXT_PUBLIC_* are inlined into the client bundle at build time. The + # browser talks to the backend over the published host port, so use + # localhost here rather than the in-network service name. + NEXT_PUBLIC_API_URL: http://localhost:3001/v1 + container_name: flowfi-frontend + environment: + NODE_ENV: production + PORT: 3000 + ports: + - "3000:3000" + depends_on: + backend: + condition: service_healthy + healthcheck: + test: ["CMD", "wget", "--no-verbose", "--tries=1", "--spider", "http://127.0.0.1:3000/health"] + interval: 15s + timeout: 5s + retries: 5 + start_period: 20s restart: unless-stopped volumes: diff --git a/frontend/Dockerfile b/frontend/Dockerfile new file mode 100644 index 00000000..d6ca2739 --- /dev/null +++ b/frontend/Dockerfile @@ -0,0 +1,69 @@ +# ============================================================================ +# Stage 1 — builder +# Installs the npm workspace from the repo root, then runs `next build`, which +# emits a self-contained server bundle (`output: "standalone"` in +# frontend/next.config.ts) instead of the full `.next` tree + node_modules. +# ============================================================================ +FROM node:20-alpine@sha256:fb4cd12c85ee03686f6af5362a0b0d56d50c58a04632e6c0fb8363f609372293 AS builder + +WORKDIR /app + +# NEXT_PUBLIC_* values are inlined into the client bundle at build time, so +# they must be passed as build args (runtime env has no effect on them). +# Defaults target the local Docker Compose stack; override with `--build-arg` +# for a real deployment. +ARG NEXT_PUBLIC_API_URL=http://localhost:3001/v1 +ARG NEXT_PUBLIC_STELLAR_NETWORK=TESTNET +ARG NEXT_PUBLIC_SOROBAN_RPC_URL=https://soroban-testnet.stellar.org +ARG NEXT_PUBLIC_NETWORK_PASSPHRASE="Test SDF Network ; September 2015" +ENV NEXT_PUBLIC_API_URL=$NEXT_PUBLIC_API_URL \ + NEXT_PUBLIC_STELLAR_NETWORK=$NEXT_PUBLIC_STELLAR_NETWORK \ + NEXT_PUBLIC_SOROBAN_RPC_URL=$NEXT_PUBLIC_SOROBAN_RPC_URL \ + NEXT_PUBLIC_NETWORK_PASSPHRASE=$NEXT_PUBLIC_NETWORK_PASSPHRASE + +# Copy the workspace manifests first so the (slow) dependency layer stays +# cached until a package.json or the root lockfile actually changes. +COPY package.json package-lock.json ./ +COPY frontend/package.json ./frontend/ +COPY backend/package.json ./backend/ + +# `npm ci` is the reproducible path; fall back to `npm install` when the +# committed lockfile and package.json are out of sync so the image still +# builds (dependency drift is tracked separately from this issue). +RUN npm ci --no-audit --no-fund || npm install --no-audit --no-fund + +COPY frontend ./frontend + +RUN npm run build --workspace frontend + +# ============================================================================ +# Stage 2 — production runner +# Copies only the traced standalone output (never the workspace node_modules) +# and runs as the non-root `node` user that ships with the Alpine base image. +# ============================================================================ +FROM node:20-alpine@sha256:fb4cd12c85ee03686f6af5362a0b0d56d50c58a04632e6c0fb8363f609372293 AS runner + +WORKDIR /app + +ENV NODE_ENV=production \ + NEXT_TELEMETRY_DISABLED=1 \ + PORT=3000 \ + HOSTNAME=0.0.0.0 + +# The standalone bundle mirrors the monorepo layout, so the entrypoint is +# `frontend/server.js` and that prefix must be preserved. `.next/static` and +# `public/` are not part of the standalone output and are copied explicitly. +COPY --from=builder --chown=node:node /app/frontend/.next/standalone ./ +COPY --from=builder --chown=node:node /app/frontend/.next/static ./frontend/.next/static +COPY --from=builder --chown=node:node /app/frontend/public ./frontend/public + +USER node + +EXPOSE 3000 + +# Native healthcheck against the app's own /health route handler so Compose / +# Kubernetes can gate traffic on readiness, not just process liveness. +HEALTHCHECK --interval=15s --timeout=5s --start-period=20s --retries=5 \ + CMD wget --no-verbose --tries=1 --spider http://127.0.0.1:3000/health || exit 1 + +CMD ["node", "frontend/server.js"] diff --git a/frontend/next.config.ts b/frontend/next.config.ts index c90d14b9..c4c76af9 100644 --- a/frontend/next.config.ts +++ b/frontend/next.config.ts @@ -2,6 +2,11 @@ import path from "node:path"; import type { NextConfig } from "next"; const nextConfig: NextConfig = { + // Emit a self-contained server bundle (`.next/standalone`) carrying only the + // traced production dependencies. The production Docker image copies this + // instead of the full `.next` tree + node_modules, keeping the runtime image + // small (issue #1335). + output: "standalone", // The workspace root lives one level above this directory. Pinning it here // prevents Turbopack from inferring a wrong root when stray package-lock // files exist outside the repo (e.g. ~/package-lock.json). diff --git a/frontend/src/app/health/route.ts b/frontend/src/app/health/route.ts new file mode 100644 index 00000000..6c908177 --- /dev/null +++ b/frontend/src/app/health/route.ts @@ -0,0 +1,8 @@ +/** + * Lightweight liveness endpoint used by the container HEALTHCHECK + * (issue #1335). Intentionally dependency-free so it stays fast and always + * available, even while other routes are still warming up. + */ +export function GET() { + return Response.json({ status: "ok" }); +} From b221f51d55f4e5db73a76e12575ed02711486747 Mon Sep 17 00:00:00 2001 From: nemeregiftbrown-byte Date: Mon, 5 Oct 2026 09:53:05 +0000 Subject: [PATCH 2/4] fix(docker): pre-fetch Prisma schema engine in backend image MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The hardened runner installs production deps with --ignore-scripts, which skips the @prisma/engines postinstall, so the Prisma schema-engine binary was missing and `prisma db push` tried to download it at runtime into root-owned node_modules. The container runs as the non-root `node` user, so the migration step in the Backend Docker Image CI job failed with "Can't write to /app/node_modules/@prisma/engines". Run the engines postinstall explicitly as root at build time so the binary ships in the image and the runtime user only reads it, preserving the non-root, production-only hardening. 🤖 Generated with Codebuff Co-Authored-By: Codebuff --- backend/Dockerfile | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/backend/Dockerfile b/backend/Dockerfile index 11f424cb..59ea71e8 100644 --- a/backend/Dockerfile +++ b/backend/Dockerfile @@ -44,6 +44,18 @@ COPY package*.json ./ RUN npm install --omit=dev --ignore-scripts --no-audit --no-fund \ && npm cache clean --force +# `--ignore-scripts` also skips the @prisma/engines postinstall, so the +# Prisma schema-engine binary is absent from the image. `prisma db push` +# (run by the CI health-check step and by anyone applying migrations in the +# container) would then try to download it at runtime into the root-owned +# node_modules tree, which the non-root `node` user cannot write to: +# Error: Can't write to /app/node_modules/@prisma/engines +# Run the engines postinstall explicitly here, as root, while the layer is +# being built so the binary ships in the image and the runtime user only ever +# reads it. The @prisma/client postinstall stays skipped (its generated client +# is copied from the builder), so no schema is required at this point. +RUN node /app/node_modules/@prisma/engines/scripts/postinstall.js + # Copy the prisma schema into the runner. The boot-and-check-health step # in .github/workflows/ci.yml runs `npx prisma db push` against this # container; without prisma/schema.prisma on disk, that command errors From 2e7a1304cf266bf5ef2e75c7b6a53a81814a0b98 Mon Sep 17 00:00:00 2001 From: nemeregiftbrown-byte Date: Wed, 7 Oct 2026 12:06:37 +0000 Subject: [PATCH 3/4] fix(docker): repair bad merge that duplicated frontend healthcheck keys MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The main->branch merge (7535117) left docker-compose.yml invalid YAML: main's node-based backend healthcheck was appended to the frontend service, so the frontend healthcheck carried duplicate `test`, `interval`, `timeout`, `retries` and `start_period` keys. `docker compose build` then failed to parse the file and Backend Docker Image CI died at the Docker Compose Build step before any image was built. Restore the backend healthcheck to main's node-based form (the merge had reverted it to the older wget version) and leave the frontend with a single wget healthcheck on port 3000, so merging both parents yields one valid, working healthcheck per service. 🤖 Generated with Codebuff Co-Authored-By: Codebuff --- docker-compose.yml | 25 ++++++++----------------- 1 file changed, 8 insertions(+), 17 deletions(-) diff --git a/docker-compose.yml b/docker-compose.yml index 0833fc8b..c6cb9c78 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -136,15 +136,17 @@ services: mock-soroban-rpc: condition: service_healthy healthcheck: - # Probe 127.0.0.1 explicitly: busybox wget resolves `localhost` to IPv6 - # first, which the IPv4-bound server does not answer. - test: ["CMD", "wget", "--no-verbose", "--tries=1", "--spider", "http://127.0.0.1:3001/health"] + # node-based check: the runtime image only ships busybox wget, which does + # not understand GNU flags like --no-verbose/--tries. + test: + - CMD + - node + - "-e" + - "require('http').get('http://127.0.0.1:3001/health', (res) => process.exit(res.statusCode === 200 ? 0 : 1)).on('error', () => process.exit(1))" interval: 10s timeout: 5s retries: 5 - # Prisma + DB warm-up happens after the port opens; give it room before - # the first probe is counted against the retry budget. - start_period: 15s + start_period: 30s restart: unless-stopped frontend: @@ -171,17 +173,6 @@ services: timeout: 5s retries: 5 start_period: 20s - # node-based check: the runtime image only ships busybox wget, which does - # not understand GNU flags like --no-verbose/--tries. - test: - - CMD - - node - - "-e" - - "require('http').get('http://127.0.0.1:3001/health', (res) => process.exit(res.statusCode === 200 ? 0 : 1)).on('error', () => process.exit(1))" - interval: 10s - timeout: 5s - retries: 5 - start_period: 30s restart: unless-stopped volumes: From fbfa3be7eda8db530bbb1c0a556871f3f8ce9668 Mon Sep 17 00:00:00 2001 From: nemeregiftbrown-byte Date: Wed, 7 Oct 2026 12:28:32 +0000 Subject: [PATCH 4/4] fix(docker): skip Prisma client regen in backend Docker CI (#1335) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The hardened backend image runs as the non-root `node` user, so the CI health-check's `prisma db push` failed when its default post-step `prisma generate` tried to write engines into the root-owned node_modules tree ("Can't write to /app/node_modules/prisma"). The generated client is already baked into dist/generated by the builder stage, so pass --skip-generate to run only the schema push. 🤖 Generated with Codebuff Co-Authored-By: Codebuff --- .github/workflows/ci.yml | 8 +++++++- backend/Dockerfile | 5 ++++- 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c6f5ce86..57fc2190 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -181,7 +181,13 @@ jobs: run: | docker compose up -d postgres sleep 10 - docker compose run --rm -e DATABASE_URL=postgresql://flowfi:flowfi_dev_password@postgres:5432/flowfi backend npx -y prisma db push --accept-data-loss + # The production image runs as the non-root `node` user and already + # ships the generated Prisma client (built in the builder stage and + # copied to dist/generated). `db push` otherwise triggers `prisma + # generate`, which needs to write into the root-owned node_modules + # tree and fails with "Can't write to /app/node_modules/prisma". + # Skip it: the client is already present, so only the schema push runs. + docker compose run --rm -e DATABASE_URL=postgresql://flowfi:flowfi_dev_password@postgres:5432/flowfi backend npx -y prisma db push --accept-data-loss --skip-generate docker compose up -d backend # Poll /health instead of a fixed sleep so cold-boot / Prisma-warmup # latency does not flake the Backend Docker Image CI. diff --git a/backend/Dockerfile b/backend/Dockerfile index 59ea71e8..cccd7d54 100644 --- a/backend/Dockerfile +++ b/backend/Dockerfile @@ -59,7 +59,10 @@ RUN node /app/node_modules/@prisma/engines/scripts/postinstall.js # Copy the prisma schema into the runner. The boot-and-check-health step # in .github/workflows/ci.yml runs `npx prisma db push` against this # container; without prisma/schema.prisma on disk, that command errors -# out and the 60-iteration /health poll times the job out. +# out and the 60-iteration /health poll times the job out. That step passes +# `--skip-generate` because the client is already baked into dist/generated +# above — regenerating would need write access to the root-owned +# node_modules tree, which the non-root `node` user deliberately does not have. COPY prisma ./prisma COPY --from=builder --chown=node:node /app/dist ./dist