Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions backend/.env.example
Original file line number Diff line number Diff line change
Expand Up @@ -121,6 +121,12 @@ SANDBOX_ALLOW_QUERY_PARAM=true
# Header name used to trigger sandbox mode (default: X-Sandbox-Mode)
SANDBOX_HEADER_NAME=X-Sandbox-Mode

# ─── Health Check ─────────────────────────────────────────────────────────────
# Upper bound in milliseconds for each dependency probe in GET /health (DB,
# Redis, Soroban RPC). Probes run in parallel, so this is roughly the worst-case
# response time. Keep it below your orchestrator's probe timeout (default: 800)
HEALTHCHECK_TIMEOUT_MS=800

# ─── Observability ────────────────────────────────────────────────────────────
# Prometheus scrape endpoint (GET /metrics).
# Deny-by-default in production: the endpoint returns 404 unless at least one of
Expand Down
17 changes: 14 additions & 3 deletions backend/src/config/swagger.ts
Original file line number Diff line number Diff line change
Expand Up @@ -444,8 +444,19 @@ See [Sandbox Mode Documentation](../docs/SANDBOX_MODE.md) for details.`,
type: 'object',
required: ['status', 'db', 'indexerEnabled', 'uptime', 'checks'],
properties: {
status: { type: 'string', enum: ['ok', 'degraded'], example: 'ok' },
status: {
type: 'string',
enum: ['ok', 'degraded'],
example: 'ok',
description: '`degraded` with HTTP 503 when a liveness check fails; `degraded` with HTTP 200 when only Redis is unavailable or timed out',
},
db: { type: 'string', enum: ['connected', 'disconnected'], example: 'connected' },
redis: {
type: 'string',
enum: ['ok', 'unavailable', 'timeout', 'not_configured'],
example: 'ok',
description: 'Same as checks.redis.status',
},
indexerEnabled: { type: 'boolean', description: 'Whether the event indexer is configured' },
indexerLag: { type: 'integer', nullable: true, description: 'Seconds since last indexer update, or null when no state row exists yet' },
eventsProcessed: { type: 'integer', description: 'Lifetime count of successfully processed indexer events' },
Expand All @@ -459,7 +470,7 @@ See [Sandbox Mode Documentation](../docs/SANDBOX_MODE.md) for details.`,
properties: {
database: {
type: 'object',
properties: { status: { type: 'string', enum: ['ok', 'down'] } },
properties: { status: { type: 'string', enum: ['ok', 'down', 'timeout'] } },
},
indexer: {
type: 'object',
Expand All @@ -471,7 +482,7 @@ See [Sandbox Mode Documentation](../docs/SANDBOX_MODE.md) for details.`,
},
redis: {
type: 'object',
properties: { status: { type: 'string', enum: ['ok', 'unavailable', 'not_configured'] } },
properties: { status: { type: 'string', enum: ['ok', 'unavailable', 'timeout', 'not_configured'] } },
},
sorobanRpc: {
type: 'object',
Expand Down
41 changes: 41 additions & 0 deletions backend/src/lib/with-timeout.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,41 @@
/**
* Generic promise timeout helper.
*
* `withTimeout` races a promise against a timer and rejects with a
* `TimeoutError` if the timer wins. It only stops *waiting*: the underlying
* operation keeps running, so callers should still configure their own client
* timeouts where the library supports them.
*/
export class TimeoutError extends Error {
readonly label: string;
readonly timeoutMs: number;

constructor(label: string, timeoutMs: number) {
super(`${label} timed out after ${timeoutMs}ms`);
this.name = 'TimeoutError';
this.label = label;
this.timeoutMs = timeoutMs;
}
}

export async function withTimeout<T>(
operation: PromiseLike<T>,
timeoutMs: number,
label = 'operation',
): Promise<T> {
const pending = Promise.resolve(operation);
// If the timer wins, `pending` may still reject later. Attach a no-op
// handler so that late rejection is never reported as unhandled.
pending.catch(() => undefined);

let timer: ReturnType<typeof setTimeout> | undefined;
const timedOut = new Promise<never>((_, reject) => {
timer = setTimeout(() => reject(new TimeoutError(label, timeoutMs)), timeoutMs);
});

try {
return await Promise.race([pending, timedOut]);
} finally {
clearTimeout(timer);
}
}
202 changes: 141 additions & 61 deletions backend/src/routes/health.routes.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,12 +2,105 @@ import { Router, type Request, type Response } from 'express';
import { prisma } from '../lib/prisma.js';
import { INDEXER_STATE_ID } from '../lib/indexer-state.js';
import { setIndexerLedgers } from '../lib/metrics.js';
import { isRedisAvailable } from '../lib/redis.js';
import { checkRpcHealth } from '../services/sorobanService.js';
import { getPublisher } from '../lib/redis.js';
import { TimeoutError, withTimeout } from '../lib/with-timeout.js';
import { checkRpcHealth, getLatestLedger } from '../services/sorobanService.js';
import { sorobanEventWorker } from '../workers/soroban-event-worker.js';
import logger from '../logger.js';

const router = Router();

/**
* Upper bound for each dependency probe. Probes run in parallel, so the whole
* response takes roughly this long in the worst case. Kept below 1,000 ms to
* leave headroom for request handling and JSON serialisation (Issue #1492).
*/
export const DEFAULT_HEALTHCHECK_TIMEOUT_MS = 800;

function getHealthcheckTimeoutMs(): number {
const parsed = Number(process.env.HEALTHCHECK_TIMEOUT_MS);
return Number.isInteger(parsed) && parsed > 0 ? parsed : DEFAULT_HEALTHCHECK_TIMEOUT_MS;
}

type DatabaseStatus = 'ok' | 'down' | 'timeout';
type RedisStatus = 'ok' | 'unavailable' | 'timeout' | 'not_configured';
type IndexerState = Awaited<ReturnType<typeof prisma.indexerState.findUnique>>;

// Last reported status per component, so failures are logged once when they
// start (and once on recovery) instead of on every probe.
const lastStatus = new Map<string, string>();

function reportStatus(component: string, status: string, healthy: boolean): void {
const previous = lastStatus.get(component);
lastStatus.set(component, status);
if (previous === status || (previous === undefined && healthy)) return;

if (healthy) {
logger.info(`[Health] ${component} recovered`, { status, previous });
} else {
logger.warn(`[Health] ${component} check failing`, { status, previous: previous ?? null });
}
}

function errorMessage(err: unknown): string {
return err instanceof Error ? err.message : String(err);
}

async function checkDatabase(timeoutMs: number): Promise<DatabaseStatus> {
try {
await withTimeout(prisma.$queryRaw`SELECT 1`, timeoutMs, 'database ping');
return 'ok';
} catch (err) {
logger.debug('[Health] database ping failed', { error: errorMessage(err) });
return err instanceof TimeoutError ? 'timeout' : 'down';
}
}

async function checkRedis(timeoutMs: number): Promise<RedisStatus> {
// Redis is optional: without REDIS_URL the SSE layer runs in single-instance mode.
if (!process.env.REDIS_URL) return 'not_configured';

// Fast path: a client that is not connected cannot answer, so don't wait on it.
const client = getPublisher();
if (!client || client.status !== 'ready') return 'unavailable';

try {
await withTimeout(client.ping(), timeoutMs, 'redis ping');
return 'ok';
} catch (err) {
logger.debug('[Health] redis ping failed', { error: errorMessage(err) });
return err instanceof TimeoutError ? 'timeout' : 'unavailable';
}
}

async function readIndexerState(timeoutMs: number): Promise<IndexerState> {
try {
return await withTimeout(
prisma.indexerState.findUnique({ where: { id: INDEXER_STATE_ID } }),
timeoutMs,
'indexer state lookup',
);
} catch {
return null;
}
}

async function checkSorobanRpc(timeoutMs: number): Promise<boolean> {
try {
return await withTimeout(checkRpcHealth(timeoutMs), timeoutMs, 'soroban rpc health');
} catch {
return false;
}
}

async function readNetworkLedger(timeoutMs: number): Promise<number> {
try {
return await withTimeout(getLatestLedger(), timeoutMs, 'latest ledger lookup');
} catch {
return 0;
}
}

/**
* @openapi
* /health:
Expand All @@ -28,9 +121,15 @@ const router = Router();
* enabled and recent per-event failures spike (≥50% of attempts in the
* last 5 minutes, with ≥3 samples), the endpoint returns 503 even if
* lag looks healthy (the IndexerState upsert bumps updatedAt every poll).
* **Redis** is optional. When it is configured but unavailable or does not
* answer a ping in time, `status` is `degraded` but the response stays
* 200, so a Redis outage does not fail liveness probes.
* Every dependency probe runs in parallel and is bounded by
* `HEALTHCHECK_TIMEOUT_MS` (default 800 ms), so the endpoint answers in
* under a second even when a dependency hangs.
* responses:
* 200:
* description: Service is healthy
* description: Service is healthy (Redis may still be degraded)
* content:
* application/json:
* schema:
Expand All @@ -43,91 +142,72 @@ const router = Router();
* $ref: '#/components/schemas/HealthResponse'
*/
router.get('/', async (_req: Request, res: Response) => {
let dbStatus = 'connected';
try {
await prisma.$queryRaw`SELECT 1`;
} catch {
dbStatus = 'disconnected';
}
const timeoutMs = getHealthcheckTimeoutMs();

// Whether the event-indexer is configured (STREAM_CONTRACT_ID must be set for it to run).
const indexerEnabled = !!process.env.STREAM_CONTRACT_ID;

let indexerLag = -1;
let state: Awaited<ReturnType<typeof prisma.indexerState.findUnique>> = null;
try {
state = await prisma.indexerState.findUnique({ where: { id: INDEXER_STATE_ID } });
if (state) {
const now = Math.floor(Date.now() / 1000);
const updatedAt = Math.floor(state.updatedAt.getTime() / 1000);
indexerLag = Math.max(0, now - updatedAt);
}
// indexerLag === -1 means no state row yet (cold start) — not an error.
} catch {
indexerLag = -1;
}
// Probes are independent, so run them together: total time is bounded by
// the slowest probe (at most timeoutMs), not the sum. None of them reject.
const [database, state, redisStatus, sorobanRpcOk, networkLedger] = await Promise.all([
checkDatabase(timeoutMs),
readIndexerState(timeoutMs),
checkRedis(timeoutMs),
checkSorobanRpc(timeoutMs),
// Resolve the network tip so ledger lag is reportable without waiting for
// the next indexer poll. Failure is non-fatal: lag degrades to null.
indexerEnabled ? readNetworkLedger(timeoutMs) : Promise.resolve(0),
]);

// Resolve the network tip so ledger lag is reportable without waiting for the
// next indexer poll. Failure is non-fatal: lag degrades to null.
let networkLedger = 0;
if (indexerEnabled) {
try {
const { getLatestLedger } = await import('../services/sorobanService.js');
networkLedger = await getLatestLedger();
} catch {
networkLedger = 0;
}
}
const dbStatus = database === 'ok' ? 'connected' : 'disconnected';

// indexerLag === -1 means no state row yet (cold start) — not an error.
const indexerLag = state
? Math.max(0, Math.floor(Date.now() / 1000) - Math.floor(state.updatedAt.getTime() / 1000))
: -1;

// 503 only when: DB is down, OR the indexer is enabled and its state row is
// stale (lag > 60). A missing state row (lag === -1) is a cold-start
// condition, not a failure, even when the indexer is enabled.
const eventCounters = sorobanEventWorker.getEventCounters();

// 503 when: DB is down, OR the indexer is enabled and its state row is stale
// (lag > 60), OR recent event-processing failures are spiking. A missing state
// row (lag === -1) is a cold-start condition, not a failure, even when the
// indexer is enabled.
// 503 when: DB is down, OR the indexer is enabled and its state row is
// stale (lag > 60), OR recent event-processing failures are spiking.
// A missing state row (lag === -1) is a cold-start condition, not a failure,
// even when the indexer is enabled.
const indexerLagDegraded = indexerEnabled && indexerLag > 60;
const indexerFailureDegraded = indexerEnabled && eventCounters.degraded;
const isHealthy =
dbStatus === 'connected' && !indexerLagDegraded && !indexerFailureDegraded;
const status = isHealthy ? 'ok' : 'degraded';

// Redis is optional (single-instance SSE mode falls back gracefully when it's
// absent), so its status never affects the top-level `isHealthy` verdict.
const redisConfigured = !!process.env.REDIS_URL;
const redisStatus = !redisConfigured
? 'not_configured'
: isRedisAvailable()
? 'ok'
: 'unavailable';

// Soroban RPC reachability is reported for observability only — it does not
// gate liveness, since a transient RPC blip shouldn't take the service down.
const sorobanRpcOk = await checkRpcHealth();
const isHealthy = database === 'ok' && !indexerLagDegraded && !indexerFailureDegraded;

// Redis never affects the HTTP status (it is optional, and failing liveness
// on a Redis outage would restart every instance), but a configured Redis
// that is down or unresponsive is surfaced as a degraded status.
const redisDegraded = redisStatus === 'unavailable' || redisStatus === 'timeout';
const status = isHealthy && !redisDegraded ? 'ok' : 'degraded';

reportStatus('database', database, database === 'ok');
reportStatus('redis', redisStatus, !redisDegraded);

// Keep the Prometheus gauges in step with what /health reports, so a scrape
// taken between poll cycles still reflects the ledger the indexer reached.
setIndexerLedgers(state?.lastLedger ?? 0, networkLedger);

res.set('Cache-Control', 'no-store');
res.status(isHealthy ? 200 : 503).json({
status,
db: dbStatus,
redis: redisStatus,
indexerEnabled,
indexerLag: indexerLag === -1 ? null : indexerLag,
eventsProcessed: eventCounters.eventsProcessed,
eventsFailed: eventCounters.eventsFailed,
lastErrorAt: eventCounters.lastErrorAt,
indexerDegraded: eventCounters.degraded,
// Ledger-level lag, which is what `flowfi_indexer_lag_ledgers` tracks.
// Null when the network tip could not be resolved.
indexerLedgerLag:
networkLedger > 0 ? Math.max(0, networkLedger - (state?.lastLedger ?? 0)) : null,
eventsProcessed: eventCounters.eventsProcessed,
eventsFailed: eventCounters.eventsFailed,
lastErrorAt: eventCounters.lastErrorAt,
indexerDegraded: eventCounters.degraded,
uptime: process.uptime(),
checks: {
database: {
status: dbStatus === 'connected' ? 'ok' : 'down',
status: database,
},
indexer: {
status: !indexerEnabled ? 'disabled' : indexerFailureDegraded || indexerLagDegraded ? 'degraded' : 'ok',
Expand Down
Loading
Loading