From a5291210caa92ec70d43067b9723ae9ea646482f Mon Sep 17 00:00:00 2001 From: AdaBliss Date: Mon, 28 Sep 2026 08:55:52 -0700 Subject: [PATCH] feat(health): add /health/live and /health/ready probes with dependency checks Add orchestrator-grade liveness and readiness probes: - GET /health/live returns 200 whenever the process is running and performs no dependency checks, so a downstream outage never triggers a restart. - GET /health/ready probes the database (SELECT 1) and cache (Redis PING) in parallel, each bounded by a 2s timeout, and returns 200 or 503 with a per-dependency status, latency and error report under `services`. - Both probes are served outside the global API prefix, like /metrics, so probe paths are stable across API versions. Make the health controller usable by load balancers: - Mark it @Public(); previously every health route required a JWT. - Exempt it from both named throttler tiers so probes cannot receive 429s. - Exclude it from the audit trail so probes do not write an audit row per request (or attempt to while the database is down). Fix the Redis indicator to probe the shared REDIS_CLIENT built from the validated REDIS_* config. It previously read an undefined REDIS_URL, always probed localhost:6379, kept its own never-closed connection, and could hang while ioredis queued the PING during an outage. Closes #351 --- API_DOCUMENTATION.md | 49 ++++++++ src/main.ts | 10 +- src/modules/health/health.controller.spec.ts | 103 +++++++++++++++ src/modules/health/health.controller.ts | 64 ++++++++++ src/modules/health/health.http.spec.ts | 117 ++++++++++++++++++ .../health/indicators/redis.health.spec.ts | 66 +++++++--- src/modules/health/indicators/redis.health.ts | 76 ++++++++---- 7 files changed, 437 insertions(+), 48 deletions(-) create mode 100644 src/modules/health/health.http.spec.ts diff --git a/API_DOCUMENTATION.md b/API_DOCUMENTATION.md index dd926d58..e460d2b3 100644 --- a/API_DOCUMENTATION.md +++ b/API_DOCUMENTATION.md @@ -372,6 +372,55 @@ Delete a budget. --- +## Health Probes (`/health`) + +The liveness and readiness probes are served **outside** the API prefix, so +orchestrator and load-balancer probe paths do not change with the API version. +Both are public, exempt from rate limiting, excluded from the audit trail, and +return raw JSON (no success envelope). + +### GET `/health/live` +Liveness probe. Returns `200` whenever the process is running. It performs no +dependency checks, so a database or cache outage never causes an otherwise +healthy process to be restarted. + +**Authentication:** Public + +**Response (200):** +```json +{ "status": "up", "timestamp": "2026-09-28T10:00:00.000Z", "uptimeSeconds": 42 } +``` + +### GET `/health/ready` +Readiness probe. Probes the database (`SELECT 1`) and cache (Redis `PING`) in +parallel, each bounded by a 2 second timeout. Returns `200` when every +dependency is up and `503` when any is down. + +**Authentication:** Public + +**Response (503 example):** +```json +{ + "status": "down", + "timestamp": "2026-09-28T10:00:00.000Z", + "services": { + "database": { + "status": "down", + "latencyMs": 2001, + "timestamp": "2026-09-28T10:00:00.000Z", + "error": "Database health check timed out after 2000ms" + }, + "cache": { "status": "up", "latencyMs": 1, "timestamp": "2026-09-28T10:00:00.000Z" } + } +} +``` + +Richer diagnostics (including Stellar and migration status) remain available +under the API prefix at `GET /{API_PREFIX}/health/readiness`, +`GET /{API_PREFIX}/health/liveness` and `GET /{API_PREFIX}/health/database`. + +--- + ## Common Types ### Pagination Query diff --git a/src/main.ts b/src/main.ts index 91df3f89..a624fe67 100644 --- a/src/main.ts +++ b/src/main.ts @@ -65,9 +65,15 @@ async function bootstrap() { // API prefix (e.g. api/v1). Versioning is expressed via this stable prefix // rather than Nest URI versioning to avoid a duplicated version segment. // `/metrics` is excluded so it stays at a fixed, unversioned path for - // Prometheus scrape configs. + // Prometheus scrape configs. The liveness/readiness probes are excluded for + // the same reason: orchestrator and load-balancer probe paths must not change + // when the API version does. app.setGlobalPrefix(appConfig.apiPrefix, { - exclude: [{ path: 'metrics', method: RequestMethod.GET }], + exclude: [ + { path: 'metrics', method: RequestMethod.GET }, + { path: 'health/live', method: RequestMethod.GET }, + { path: 'health/ready', method: RequestMethod.GET }, + ], }); // OpenAPI / Swagger documentation diff --git a/src/modules/health/health.controller.spec.ts b/src/modules/health/health.controller.spec.ts index 8992c82c..23310744 100644 --- a/src/modules/health/health.controller.spec.ts +++ b/src/modules/health/health.controller.spec.ts @@ -68,6 +68,109 @@ describe('HealthController', () => { ); }); + describe('GET /health/live', () => { + it('returns 200 with process uptime without probing any dependency', () => { + controller.live(res as Response); + + expect(res.status).toHaveBeenCalledWith(200); + expect(res.json).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'up', + timestamp: expect.any(String), + uptimeSeconds: expect.any(Number), + }), + ); + expect(dbHealth.check).not.toHaveBeenCalled(); + expect(redisHealth.checkHealth).not.toHaveBeenCalled(); + }); + + it('stays 200 during a database outage', () => { + dbHealth.check.mockRejectedValue(new Error('ECONNREFUSED')); + + controller.live(res as Response); + + expect(res.status).toHaveBeenCalledWith(200); + }); + }); + + describe('GET /health/ready', () => { + it('returns 200 with per-dependency status when database and cache are up', async () => { + await controller.ready(res as Response); + + expect(res.status).toHaveBeenCalledWith(200); + expect(res.json).toHaveBeenCalledWith({ + status: 'up', + timestamp: expect.any(String), + services: { + database: expect.objectContaining({ status: 'up', latencyMs: 5 }), + cache: expect.objectContaining({ status: 'up', latencyMs: 2 }), + }, + }); + }); + + it('returns 503 during a simulated database outage', async () => { + dbHealth.check.mockResolvedValue( + terminus({ status: 'down', error: 'Error', message: 'Database health check timed out after 2000ms' }), + ); + + await controller.ready(res as Response); + + expect(res.status).toHaveBeenCalledWith(503); + expect(res.json).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'down', + services: { + database: expect.objectContaining({ + status: 'down', + error: 'Database health check timed out after 2000ms', + }), + cache: expect.objectContaining({ status: 'up' }), + }, + }), + ); + }); + + it('returns 503 during a simulated cache outage', async () => { + redisHealth.checkHealth.mockResolvedValue({ + status: 'down', + latencyMs: 2000, + timestamp: new Date().toISOString(), + error: 'Redis health check timed out after 2000ms', + }); + + await controller.ready(res as Response); + + expect(res.status).toHaveBeenCalledWith(503); + expect(res.json).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'down', + services: expect.objectContaining({ + database: expect.objectContaining({ status: 'up' }), + cache: expect.objectContaining({ status: 'down' }), + }), + }), + ); + }); + + it('returns 503 when the database indicator produces no result', async () => { + dbHealth.check.mockResolvedValue({}); + + await controller.ready(res as Response); + + expect(res.status).toHaveBeenCalledWith(503); + }); + + it('probes only critical dependencies, so an external Stellar outage cannot fail readiness', async () => { + stellarHealth.checkHealth.mockResolvedValue({ status: 'down', timestamp: 'now' }); + + await controller.ready(res as Response); + + expect(stellarHealth.checkHealth).not.toHaveBeenCalled(); + expect(migrationHealth.checkHealth).not.toHaveBeenCalled(); + expect(res.status).toHaveBeenCalledWith(200); + }); + }); + it('returns liveness payload with status up', () => { const response = controller.getLiveness(); expect(response.status).toBe('up'); diff --git a/src/modules/health/health.controller.ts b/src/modules/health/health.controller.ts index 03156ddc..b73810c8 100644 --- a/src/modules/health/health.controller.ts +++ b/src/modules/health/health.controller.ts @@ -1,7 +1,10 @@ import { Controller, Get, Res, HttpStatus } from '@nestjs/common'; import { ApiOperation, ApiResponse, ApiTags } from '@nestjs/swagger'; import { HealthIndicatorResult } from '@nestjs/terminus'; +import { SkipThrottle } from '@nestjs/throttler'; import { Response } from 'express'; +import { Public } from '../../common/decorators/public.decorator'; +import { SkipAudit } from '../../common/decorators/skip-audit.decorator'; import { PrismaHealthIndicator } from './indicators/prisma.health'; import { RedisHealthIndicator } from './indicators/redis.health'; import { StellarHealthIndicator } from './indicators/stellar.health'; @@ -14,8 +17,21 @@ interface ReadinessServiceReport { [key: string]: unknown; } +/** + * Health and probe endpoints. Public (orchestrators and load balancers carry no + * credentials), excluded from rate limiting so frequent probes can never be + * answered with a 429, and excluded from the audit trail so probes do not write + * a row per request — or attempt to while the database is down. + * + * `GET /health/live` and `GET /health/ready` are the orchestrator probes and are + * served outside the global API prefix (see `main.ts`); the remaining routes are + * richer diagnostics served under it. + */ @ApiTags('Health') @Controller('health') +@Public() +@SkipAudit() +@SkipThrottle({ api: true, auth: true }) export class HealthController { constructor( private readonly dbIndicator: PrismaHealthIndicator, @@ -24,6 +40,54 @@ export class HealthController { private readonly migrationIndicator: DatabaseMigrationHealthIndicator, ) {} + @Get('live') + @ApiOperation({ + summary: 'Liveness probe', + description: + 'Returns 200 whenever the process is running and able to serve HTTP. Performs no ' + + 'dependency checks, so a downstream outage never causes the orchestrator to restart ' + + 'an otherwise healthy process.', + }) + @ApiResponse({ status: 200, description: 'Process is alive' }) + live(@Res() res: Response) { + return res.status(HttpStatus.OK).json({ + status: 'up', + timestamp: new Date().toISOString(), + uptimeSeconds: Math.floor(process.uptime()), + }); + } + + @Get('ready') + @ApiOperation({ + summary: 'Readiness probe', + description: + 'Probes the critical dependencies (database and cache) in parallel. Returns 200 when ' + + 'every dependency is up and 503 when any is down, with per-dependency status, latency ' + + 'and error detail under `services`.', + }) + @ApiResponse({ status: 200, description: 'All critical dependencies are reachable' }) + @ApiResponse({ status: 503, description: 'At least one critical dependency is unreachable' }) + async ready(@Res() res: Response) { + const [database, cache] = await Promise.all([ + this.dbIndicator.check('database'), + this.redisIndicator.checkHealth(), + ]); + + const services = { + database: unwrap(database, 'database'), + cache, + }; + + const isReady = Object.values(services).every((s) => s.status === 'up'); + const statusCode = isReady ? HttpStatus.OK : HttpStatus.SERVICE_UNAVAILABLE; + + return res.status(statusCode).json({ + status: isReady ? 'up' : 'down', + timestamp: new Date().toISOString(), + services, + }); + } + @Get('liveness') @ApiOperation({ summary: 'Application liveness check' }) @ApiResponse({ status: 200, description: 'Application is alive' }) diff --git a/src/modules/health/health.http.spec.ts b/src/modules/health/health.http.spec.ts new file mode 100644 index 00000000..dd2633a5 --- /dev/null +++ b/src/modules/health/health.http.spec.ts @@ -0,0 +1,117 @@ +import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'; +import { INestApplication, RequestMethod } from '@nestjs/common'; +import { APP_GUARD } from '@nestjs/core'; +import { Test } from '@nestjs/testing'; +import { AddressInfo } from 'net'; +import { JwtAuthGuard } from '../../common/guards/jwt-auth.guard'; +import { HealthController } from './health.controller'; +import { PrismaHealthIndicator } from './indicators/prisma.health'; +import { RedisHealthIndicator } from './indicators/redis.health'; +import { StellarHealthIndicator } from './indicators/stellar.health'; +import { DatabaseMigrationHealthIndicator } from './indicators/database-migration.health'; + +/** + * HTTP-level coverage for the orchestrator probes: real routing, the global + * authentication guard, and the same prefix exclusions `main.ts` applies. The + * indicators are stubbed so dependency outages can be simulated + * deterministically. + */ +describe('Health probes over HTTP', () => { + let app: INestApplication; + let baseUrl: string; + + const dbIndicator = { check: vi.fn() }; + const redisIndicator = { checkHealth: vi.fn() }; + + const databaseUp = () => ({ + database: { status: 'up', latencyMs: 3, timestamp: new Date().toISOString() }, + }); + const cacheUp = () => ({ status: 'up', latencyMs: 1, timestamp: new Date().toISOString() }); + + beforeAll(async () => { + const moduleRef = await Test.createTestingModule({ + controllers: [HealthController], + providers: [ + { provide: APP_GUARD, useClass: JwtAuthGuard }, + { provide: PrismaHealthIndicator, useValue: dbIndicator }, + { provide: RedisHealthIndicator, useValue: redisIndicator }, + { provide: StellarHealthIndicator, useValue: { checkHealth: vi.fn() } }, + { provide: DatabaseMigrationHealthIndicator, useValue: { checkHealth: vi.fn() } }, + ], + }).compile(); + + app = moduleRef.createNestApplication({ logger: false }); + app.setGlobalPrefix('api/v1', { + exclude: [ + { path: 'health/live', method: RequestMethod.GET }, + { path: 'health/ready', method: RequestMethod.GET }, + ], + }); + await app.listen(0, '127.0.0.1'); + const { port } = app.getHttpServer().address() as AddressInfo; + baseUrl = `http://127.0.0.1:${port}`; + }); + + afterAll(async () => { + await app.close(); + }); + + beforeEach(() => { + dbIndicator.check.mockReset().mockResolvedValue(databaseUp()); + redisIndicator.checkHealth.mockReset().mockResolvedValue(cacheUp()); + }); + + it('serves GET /health/live without credentials and outside the API prefix', async () => { + const res = await fetch(`${baseUrl}/health/live`); + + expect(res.status).toBe(200); + expect(await res.json()).toMatchObject({ status: 'up' }); + }); + + it('serves GET /health/ready without credentials with 200 when dependencies are up', async () => { + const res = await fetch(`${baseUrl}/health/ready`); + + expect(res.status).toBe(200); + expect(await res.json()).toMatchObject({ + status: 'up', + services: { database: { status: 'up' }, cache: { status: 'up' } }, + }); + }); + + it('answers GET /health/ready with 503 and structured detail during a database outage', async () => { + dbIndicator.check.mockResolvedValue({ + database: { + status: 'down', + latencyMs: 2000, + timestamp: new Date().toISOString(), + error: 'Error', + message: 'Database health check timed out after 2000ms', + }, + }); + + const res = await fetch(`${baseUrl}/health/ready`); + + expect(res.status).toBe(503); + expect(await res.json()).toMatchObject({ + status: 'down', + services: { + database: { status: 'down', error: 'Database health check timed out after 2000ms' }, + cache: { status: 'up' }, + }, + }); + }); + + it('keeps GET /health/live at 200 during a database outage', async () => { + dbIndicator.check.mockResolvedValue({ database: { status: 'down' } }); + + const res = await fetch(`${baseUrl}/health/live`); + + expect(res.status).toBe(200); + }); + + it('keeps the diagnostic routes under the API prefix and public', async () => { + const res = await fetch(`${baseUrl}/api/v1/health/database`); + + expect(res.status).toBe(200); + }); +}); diff --git a/src/modules/health/indicators/redis.health.spec.ts b/src/modules/health/indicators/redis.health.spec.ts index 22cf89cd..80fd3753 100644 --- a/src/modules/health/indicators/redis.health.spec.ts +++ b/src/modules/health/indicators/redis.health.spec.ts @@ -1,44 +1,70 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import Redis from 'ioredis'; import { RedisHealthIndicator } from './redis.health'; -import { ConfigService } from '@nestjs/config'; - -const mockPing = vi.fn(); -vi.mock('ioredis', () => { - return { - default: vi.fn().mockImplementation(() => ({ - ping: mockPing, - })), - }; -}); describe('RedisHealthIndicator', () => { - let configService: Partial; + let redis: { status: string; ping: ReturnType }; let indicator: RedisHealthIndicator; beforeEach(() => { - vi.clearAllMocks(); - configService = { - get: vi.fn().mockReturnValue('redis://localhost:6379'), - }; - indicator = new RedisHealthIndicator(configService as ConfigService); + redis = { status: 'ready', ping: vi.fn() }; + indicator = new RedisHealthIndicator(redis as unknown as Redis); + }); + + afterEach(() => { + vi.useRealTimers(); }); it('returns UP when ping returns PONG', async () => { - mockPing.mockResolvedValue('PONG'); + redis.ping.mockResolvedValue('PONG'); const report = await indicator.checkHealth(); + expect(redis.ping).toHaveBeenCalledTimes(1); expect(report.status).toBe('up'); - expect(report.latencyMs).toBeDefined(); + expect(report.latencyMs).toBeGreaterThanOrEqual(0); expect(report.error).toBeUndefined(); }); it('returns DOWN when ping fails', async () => { - mockPing.mockRejectedValue(new Error('Redis connection refused')); + redis.ping.mockRejectedValue(new Error('Redis connection refused')); const report = await indicator.checkHealth(); expect(report.status).toBe('down'); expect(report.error).toContain('Redis connection refused'); }); + + it('returns DOWN on an unexpected ping reply', async () => { + redis.ping.mockResolvedValue('LOADING'); + + const report = await indicator.checkHealth(); + + expect(report.status).toBe('down'); + expect(report.error).toContain('Unexpected ping response'); + }); + + it('returns DOWN without pinging when the client connection has ended', async () => { + redis.status = 'end'; + + const report = await indicator.checkHealth(); + + expect(redis.ping).not.toHaveBeenCalled(); + expect(report.status).toBe('down'); + expect(report.error).toContain('closed'); + }); + + it('returns DOWN once the probe exceeds its timeout', async () => { + vi.useFakeTimers(); + // Simulates ioredis holding the command in its offline queue while Redis is + // unreachable: the promise never settles on its own. + redis.ping.mockReturnValue(new Promise(() => undefined)); + + const pending = indicator.checkHealth(500); + await vi.advanceTimersByTimeAsync(500); + const report = await pending; + + expect(report.status).toBe('down'); + expect(report.error).toContain('timed out after 500ms'); + }); }); diff --git a/src/modules/health/indicators/redis.health.ts b/src/modules/health/indicators/redis.health.ts index 4858a982..b1b1bbdc 100644 --- a/src/modules/health/indicators/redis.health.ts +++ b/src/modules/health/indicators/redis.health.ts @@ -1,6 +1,6 @@ -import { Injectable, Logger } from '@nestjs/common'; -import { ConfigService } from '@nestjs/config'; +import { Inject, Injectable, Logger } from '@nestjs/common'; import Redis from 'ioredis'; +import { REDIS_CLIENT } from '../../../common/locks/locks.constants'; export interface RedisHealthReport { status: 'up' | 'down'; @@ -9,37 +9,38 @@ export interface RedisHealthReport { error?: string; } +/** + * Probes the cache store by issuing a `PING` on the application's shared Redis + * client (provided by `LocksModule` from the validated `REDIS_*` config), so the + * check exercises the exact connection the API depends on rather than a + * side-channel client pointed at a default host. + * + * Like the database indicator, the probe is: + * - **Bounded.** ioredis queues commands while reconnecting, so an unreachable + * Redis would otherwise hold the probe open until retries are exhausted. + * - **Never throws.** Any failure is reported as `status: 'down'`. + */ @Injectable() export class RedisHealthIndicator { private readonly logger = new Logger(RedisHealthIndicator.name); - private redisClient: Redis | null = null; - constructor(private readonly configService: ConfigService) {} + /** Ceiling on a single probe, in ms. */ + static readonly DEFAULT_TIMEOUT_MS = 2_000; - private getClient(): Redis { - if (!this.redisClient) { - try { - const redisUrl = this.configService.get('REDIS_URL') || process.env.REDIS_URL || 'redis://localhost:6379'; - this.redisClient = new Redis(redisUrl, { - lazyConnect: true, - enableReadyCheck: true, - maxRetriesPerRequest: 1, - }); - } catch (err) { - this.logger.warn(`Failed to initialize Redis client for health check: ${err}`); - } - } - if (!this.redisClient) { - throw new Error('Redis client could not be initialized'); - } - return this.redisClient; - } + constructor(@Inject(REDIS_CLIENT) private readonly redis: Redis) {} - async checkHealth(): Promise { + async checkHealth( + timeoutMs: number = RedisHealthIndicator.DEFAULT_TIMEOUT_MS, + ): Promise { const start = Date.now(); try { - const client = this.getClient(); - const res = await client.ping(); + // A client that has been explicitly closed will never reconnect; fail + // fast instead of waiting for the timeout. + if (this.redis.status === 'end') { + throw new Error('Redis connection is closed'); + } + + const res = await this.withTimeout(this.redis.ping(), timeoutMs); const latencyMs = Date.now() - start; if (res !== 'PONG') { @@ -54,7 +55,7 @@ export class RedisHealthIndicator { } catch (error) { const latencyMs = Date.now() - start; const message = error instanceof Error ? error.message : String(error); - this.logger.error(`Redis health check failed: ${message}`); + this.logger.error(`Redis health check failed after ${latencyMs}ms: ${message}`); return { status: 'down', @@ -64,4 +65,27 @@ export class RedisHealthIndicator { }; } } + + private withTimeout(probe: Promise, timeoutMs: number): Promise { + if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) { + return probe; + } + + return new Promise((resolve, reject) => { + const timer = setTimeout(() => { + reject(new Error(`Redis health check timed out after ${timeoutMs}ms`)); + }, timeoutMs); + + probe.then( + (value) => { + clearTimeout(timer); + resolve(value); + }, + (error: unknown) => { + clearTimeout(timer); + reject(error instanceof Error ? error : new Error(String(error))); + }, + ); + }); + } }