diff --git a/API_DOCUMENTATION.md b/API_DOCUMENTATION.md index dd926d58..e460d2b3 100644 --- a/API_DOCUMENTATION.md +++ b/API_DOCUMENTATION.md @@ -372,6 +372,55 @@ Delete a budget. --- +## Health Probes (`/health`) + +The liveness and readiness probes are served **outside** the API prefix, so +orchestrator and load-balancer probe paths do not change with the API version. +Both are public, exempt from rate limiting, excluded from the audit trail, and +return raw JSON (no success envelope). + +### GET `/health/live` +Liveness probe. Returns `200` whenever the process is running. It performs no +dependency checks, so a database or cache outage never causes an otherwise +healthy process to be restarted. + +**Authentication:** Public + +**Response (200):** +```json +{ "status": "up", "timestamp": "2026-09-28T10:00:00.000Z", "uptimeSeconds": 42 } +``` + +### GET `/health/ready` +Readiness probe. Probes the database (`SELECT 1`) and cache (Redis `PING`) in +parallel, each bounded by a 2 second timeout. Returns `200` when every +dependency is up and `503` when any is down. + +**Authentication:** Public + +**Response (503 example):** +```json +{ + "status": "down", + "timestamp": "2026-09-28T10:00:00.000Z", + "services": { + "database": { + "status": "down", + "latencyMs": 2001, + "timestamp": "2026-09-28T10:00:00.000Z", + "error": "Database health check timed out after 2000ms" + }, + "cache": { "status": "up", "latencyMs": 1, "timestamp": "2026-09-28T10:00:00.000Z" } + } +} +``` + +Richer diagnostics (including Stellar and migration status) remain available +under the API prefix at `GET /{API_PREFIX}/health/readiness`, +`GET /{API_PREFIX}/health/liveness` and `GET /{API_PREFIX}/health/database`. + +--- + ## Common Types ### Pagination Query diff --git a/src/main.ts b/src/main.ts index 91df3f89..a624fe67 100644 --- a/src/main.ts +++ b/src/main.ts @@ -65,9 +65,15 @@ async function bootstrap() { // API prefix (e.g. api/v1). Versioning is expressed via this stable prefix // rather than Nest URI versioning to avoid a duplicated version segment. // `/metrics` is excluded so it stays at a fixed, unversioned path for - // Prometheus scrape configs. + // Prometheus scrape configs. The liveness/readiness probes are excluded for + // the same reason: orchestrator and load-balancer probe paths must not change + // when the API version does. app.setGlobalPrefix(appConfig.apiPrefix, { - exclude: [{ path: 'metrics', method: RequestMethod.GET }], + exclude: [ + { path: 'metrics', method: RequestMethod.GET }, + { path: 'health/live', method: RequestMethod.GET }, + { path: 'health/ready', method: RequestMethod.GET }, + ], }); // OpenAPI / Swagger documentation diff --git a/src/modules/health/health.controller.spec.ts b/src/modules/health/health.controller.spec.ts index 8992c82c..23310744 100644 --- a/src/modules/health/health.controller.spec.ts +++ b/src/modules/health/health.controller.spec.ts @@ -68,6 +68,109 @@ describe('HealthController', () => { ); }); + describe('GET /health/live', () => { + it('returns 200 with process uptime without probing any dependency', () => { + controller.live(res as Response); + + expect(res.status).toHaveBeenCalledWith(200); + expect(res.json).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'up', + timestamp: expect.any(String), + uptimeSeconds: expect.any(Number), + }), + ); + expect(dbHealth.check).not.toHaveBeenCalled(); + expect(redisHealth.checkHealth).not.toHaveBeenCalled(); + }); + + it('stays 200 during a database outage', () => { + dbHealth.check.mockRejectedValue(new Error('ECONNREFUSED')); + + controller.live(res as Response); + + expect(res.status).toHaveBeenCalledWith(200); + }); + }); + + describe('GET /health/ready', () => { + it('returns 200 with per-dependency status when database and cache are up', async () => { + await controller.ready(res as Response); + + expect(res.status).toHaveBeenCalledWith(200); + expect(res.json).toHaveBeenCalledWith({ + status: 'up', + timestamp: expect.any(String), + services: { + database: expect.objectContaining({ status: 'up', latencyMs: 5 }), + cache: expect.objectContaining({ status: 'up', latencyMs: 2 }), + }, + }); + }); + + it('returns 503 during a simulated database outage', async () => { + dbHealth.check.mockResolvedValue( + terminus({ status: 'down', error: 'Error', message: 'Database health check timed out after 2000ms' }), + ); + + await controller.ready(res as Response); + + expect(res.status).toHaveBeenCalledWith(503); + expect(res.json).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'down', + services: { + database: expect.objectContaining({ + status: 'down', + error: 'Database health check timed out after 2000ms', + }), + cache: expect.objectContaining({ status: 'up' }), + }, + }), + ); + }); + + it('returns 503 during a simulated cache outage', async () => { + redisHealth.checkHealth.mockResolvedValue({ + status: 'down', + latencyMs: 2000, + timestamp: new Date().toISOString(), + error: 'Redis health check timed out after 2000ms', + }); + + await controller.ready(res as Response); + + expect(res.status).toHaveBeenCalledWith(503); + expect(res.json).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'down', + services: expect.objectContaining({ + database: expect.objectContaining({ status: 'up' }), + cache: expect.objectContaining({ status: 'down' }), + }), + }), + ); + }); + + it('returns 503 when the database indicator produces no result', async () => { + dbHealth.check.mockResolvedValue({}); + + await controller.ready(res as Response); + + expect(res.status).toHaveBeenCalledWith(503); + }); + + it('probes only critical dependencies, so an external Stellar outage cannot fail readiness', async () => { + stellarHealth.checkHealth.mockResolvedValue({ status: 'down', timestamp: 'now' }); + + await controller.ready(res as Response); + + expect(stellarHealth.checkHealth).not.toHaveBeenCalled(); + expect(migrationHealth.checkHealth).not.toHaveBeenCalled(); + expect(res.status).toHaveBeenCalledWith(200); + }); + }); + it('returns liveness payload with status up', () => { const response = controller.getLiveness(); expect(response.status).toBe('up'); diff --git a/src/modules/health/health.controller.ts b/src/modules/health/health.controller.ts index 03156ddc..b73810c8 100644 --- a/src/modules/health/health.controller.ts +++ b/src/modules/health/health.controller.ts @@ -1,7 +1,10 @@ import { Controller, Get, Res, HttpStatus } from '@nestjs/common'; import { ApiOperation, ApiResponse, ApiTags } from '@nestjs/swagger'; import { HealthIndicatorResult } from '@nestjs/terminus'; +import { SkipThrottle } from '@nestjs/throttler'; import { Response } from 'express'; +import { Public } from '../../common/decorators/public.decorator'; +import { SkipAudit } from '../../common/decorators/skip-audit.decorator'; import { PrismaHealthIndicator } from './indicators/prisma.health'; import { RedisHealthIndicator } from './indicators/redis.health'; import { StellarHealthIndicator } from './indicators/stellar.health'; @@ -14,8 +17,21 @@ interface ReadinessServiceReport { [key: string]: unknown; } +/** + * Health and probe endpoints. Public (orchestrators and load balancers carry no + * credentials), excluded from rate limiting so frequent probes can never be + * answered with a 429, and excluded from the audit trail so probes do not write + * a row per request — or attempt to while the database is down. + * + * `GET /health/live` and `GET /health/ready` are the orchestrator probes and are + * served outside the global API prefix (see `main.ts`); the remaining routes are + * richer diagnostics served under it. + */ @ApiTags('Health') @Controller('health') +@Public() +@SkipAudit() +@SkipThrottle({ api: true, auth: true }) export class HealthController { constructor( private readonly dbIndicator: PrismaHealthIndicator, @@ -24,6 +40,54 @@ export class HealthController { private readonly migrationIndicator: DatabaseMigrationHealthIndicator, ) {} + @Get('live') + @ApiOperation({ + summary: 'Liveness probe', + description: + 'Returns 200 whenever the process is running and able to serve HTTP. Performs no ' + + 'dependency checks, so a downstream outage never causes the orchestrator to restart ' + + 'an otherwise healthy process.', + }) + @ApiResponse({ status: 200, description: 'Process is alive' }) + live(@Res() res: Response) { + return res.status(HttpStatus.OK).json({ + status: 'up', + timestamp: new Date().toISOString(), + uptimeSeconds: Math.floor(process.uptime()), + }); + } + + @Get('ready') + @ApiOperation({ + summary: 'Readiness probe', + description: + 'Probes the critical dependencies (database and cache) in parallel. Returns 200 when ' + + 'every dependency is up and 503 when any is down, with per-dependency status, latency ' + + 'and error detail under `services`.', + }) + @ApiResponse({ status: 200, description: 'All critical dependencies are reachable' }) + @ApiResponse({ status: 503, description: 'At least one critical dependency is unreachable' }) + async ready(@Res() res: Response) { + const [database, cache] = await Promise.all([ + this.dbIndicator.check('database'), + this.redisIndicator.checkHealth(), + ]); + + const services = { + database: unwrap(database, 'database'), + cache, + }; + + const isReady = Object.values(services).every((s) => s.status === 'up'); + const statusCode = isReady ? HttpStatus.OK : HttpStatus.SERVICE_UNAVAILABLE; + + return res.status(statusCode).json({ + status: isReady ? 'up' : 'down', + timestamp: new Date().toISOString(), + services, + }); + } + @Get('liveness') @ApiOperation({ summary: 'Application liveness check' }) @ApiResponse({ status: 200, description: 'Application is alive' }) diff --git a/src/modules/health/health.http.spec.ts b/src/modules/health/health.http.spec.ts new file mode 100644 index 00000000..dd2633a5 --- /dev/null +++ b/src/modules/health/health.http.spec.ts @@ -0,0 +1,117 @@ +import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'; +import { INestApplication, RequestMethod } from '@nestjs/common'; +import { APP_GUARD } from '@nestjs/core'; +import { Test } from '@nestjs/testing'; +import { AddressInfo } from 'net'; +import { JwtAuthGuard } from '../../common/guards/jwt-auth.guard'; +import { HealthController } from './health.controller'; +import { PrismaHealthIndicator } from './indicators/prisma.health'; +import { RedisHealthIndicator } from './indicators/redis.health'; +import { StellarHealthIndicator } from './indicators/stellar.health'; +import { DatabaseMigrationHealthIndicator } from './indicators/database-migration.health'; + +/** + * HTTP-level coverage for the orchestrator probes: real routing, the global + * authentication guard, and the same prefix exclusions `main.ts` applies. The + * indicators are stubbed so dependency outages can be simulated + * deterministically. + */ +describe('Health probes over HTTP', () => { + let app: INestApplication; + let baseUrl: string; + + const dbIndicator = { check: vi.fn() }; + const redisIndicator = { checkHealth: vi.fn() }; + + const databaseUp = () => ({ + database: { status: 'up', latencyMs: 3, timestamp: new Date().toISOString() }, + }); + const cacheUp = () => ({ status: 'up', latencyMs: 1, timestamp: new Date().toISOString() }); + + beforeAll(async () => { + const moduleRef = await Test.createTestingModule({ + controllers: [HealthController], + providers: [ + { provide: APP_GUARD, useClass: JwtAuthGuard }, + { provide: PrismaHealthIndicator, useValue: dbIndicator }, + { provide: RedisHealthIndicator, useValue: redisIndicator }, + { provide: StellarHealthIndicator, useValue: { checkHealth: vi.fn() } }, + { provide: DatabaseMigrationHealthIndicator, useValue: { checkHealth: vi.fn() } }, + ], + }).compile(); + + app = moduleRef.createNestApplication({ logger: false }); + app.setGlobalPrefix('api/v1', { + exclude: [ + { path: 'health/live', method: RequestMethod.GET }, + { path: 'health/ready', method: RequestMethod.GET }, + ], + }); + await app.listen(0, '127.0.0.1'); + const { port } = app.getHttpServer().address() as AddressInfo; + baseUrl = `http://127.0.0.1:${port}`; + }); + + afterAll(async () => { + await app.close(); + }); + + beforeEach(() => { + dbIndicator.check.mockReset().mockResolvedValue(databaseUp()); + redisIndicator.checkHealth.mockReset().mockResolvedValue(cacheUp()); + }); + + it('serves GET /health/live without credentials and outside the API prefix', async () => { + const res = await fetch(`${baseUrl}/health/live`); + + expect(res.status).toBe(200); + expect(await res.json()).toMatchObject({ status: 'up' }); + }); + + it('serves GET /health/ready without credentials with 200 when dependencies are up', async () => { + const res = await fetch(`${baseUrl}/health/ready`); + + expect(res.status).toBe(200); + expect(await res.json()).toMatchObject({ + status: 'up', + services: { database: { status: 'up' }, cache: { status: 'up' } }, + }); + }); + + it('answers GET /health/ready with 503 and structured detail during a database outage', async () => { + dbIndicator.check.mockResolvedValue({ + database: { + status: 'down', + latencyMs: 2000, + timestamp: new Date().toISOString(), + error: 'Error', + message: 'Database health check timed out after 2000ms', + }, + }); + + const res = await fetch(`${baseUrl}/health/ready`); + + expect(res.status).toBe(503); + expect(await res.json()).toMatchObject({ + status: 'down', + services: { + database: { status: 'down', error: 'Database health check timed out after 2000ms' }, + cache: { status: 'up' }, + }, + }); + }); + + it('keeps GET /health/live at 200 during a database outage', async () => { + dbIndicator.check.mockResolvedValue({ database: { status: 'down' } }); + + const res = await fetch(`${baseUrl}/health/live`); + + expect(res.status).toBe(200); + }); + + it('keeps the diagnostic routes under the API prefix and public', async () => { + const res = await fetch(`${baseUrl}/api/v1/health/database`); + + expect(res.status).toBe(200); + }); +}); diff --git a/src/modules/health/indicators/redis.health.spec.ts b/src/modules/health/indicators/redis.health.spec.ts index 22cf89cd..80fd3753 100644 --- a/src/modules/health/indicators/redis.health.spec.ts +++ b/src/modules/health/indicators/redis.health.spec.ts @@ -1,44 +1,70 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import Redis from 'ioredis'; import { RedisHealthIndicator } from './redis.health'; -import { ConfigService } from '@nestjs/config'; - -const mockPing = vi.fn(); -vi.mock('ioredis', () => { - return { - default: vi.fn().mockImplementation(() => ({ - ping: mockPing, - })), - }; -}); describe('RedisHealthIndicator', () => { - let configService: Partial; + let redis: { status: string; ping: ReturnType }; let indicator: RedisHealthIndicator; beforeEach(() => { - vi.clearAllMocks(); - configService = { - get: vi.fn().mockReturnValue('redis://localhost:6379'), - }; - indicator = new RedisHealthIndicator(configService as ConfigService); + redis = { status: 'ready', ping: vi.fn() }; + indicator = new RedisHealthIndicator(redis as unknown as Redis); + }); + + afterEach(() => { + vi.useRealTimers(); }); it('returns UP when ping returns PONG', async () => { - mockPing.mockResolvedValue('PONG'); + redis.ping.mockResolvedValue('PONG'); const report = await indicator.checkHealth(); + expect(redis.ping).toHaveBeenCalledTimes(1); expect(report.status).toBe('up'); - expect(report.latencyMs).toBeDefined(); + expect(report.latencyMs).toBeGreaterThanOrEqual(0); expect(report.error).toBeUndefined(); }); it('returns DOWN when ping fails', async () => { - mockPing.mockRejectedValue(new Error('Redis connection refused')); + redis.ping.mockRejectedValue(new Error('Redis connection refused')); const report = await indicator.checkHealth(); expect(report.status).toBe('down'); expect(report.error).toContain('Redis connection refused'); }); + + it('returns DOWN on an unexpected ping reply', async () => { + redis.ping.mockResolvedValue('LOADING'); + + const report = await indicator.checkHealth(); + + expect(report.status).toBe('down'); + expect(report.error).toContain('Unexpected ping response'); + }); + + it('returns DOWN without pinging when the client connection has ended', async () => { + redis.status = 'end'; + + const report = await indicator.checkHealth(); + + expect(redis.ping).not.toHaveBeenCalled(); + expect(report.status).toBe('down'); + expect(report.error).toContain('closed'); + }); + + it('returns DOWN once the probe exceeds its timeout', async () => { + vi.useFakeTimers(); + // Simulates ioredis holding the command in its offline queue while Redis is + // unreachable: the promise never settles on its own. + redis.ping.mockReturnValue(new Promise(() => undefined)); + + const pending = indicator.checkHealth(500); + await vi.advanceTimersByTimeAsync(500); + const report = await pending; + + expect(report.status).toBe('down'); + expect(report.error).toContain('timed out after 500ms'); + }); }); diff --git a/src/modules/health/indicators/redis.health.ts b/src/modules/health/indicators/redis.health.ts index 4858a982..b1b1bbdc 100644 --- a/src/modules/health/indicators/redis.health.ts +++ b/src/modules/health/indicators/redis.health.ts @@ -1,6 +1,6 @@ -import { Injectable, Logger } from '@nestjs/common'; -import { ConfigService } from '@nestjs/config'; +import { Inject, Injectable, Logger } from '@nestjs/common'; import Redis from 'ioredis'; +import { REDIS_CLIENT } from '../../../common/locks/locks.constants'; export interface RedisHealthReport { status: 'up' | 'down'; @@ -9,37 +9,38 @@ export interface RedisHealthReport { error?: string; } +/** + * Probes the cache store by issuing a `PING` on the application's shared Redis + * client (provided by `LocksModule` from the validated `REDIS_*` config), so the + * check exercises the exact connection the API depends on rather than a + * side-channel client pointed at a default host. + * + * Like the database indicator, the probe is: + * - **Bounded.** ioredis queues commands while reconnecting, so an unreachable + * Redis would otherwise hold the probe open until retries are exhausted. + * - **Never throws.** Any failure is reported as `status: 'down'`. + */ @Injectable() export class RedisHealthIndicator { private readonly logger = new Logger(RedisHealthIndicator.name); - private redisClient: Redis | null = null; - constructor(private readonly configService: ConfigService) {} + /** Ceiling on a single probe, in ms. */ + static readonly DEFAULT_TIMEOUT_MS = 2_000; - private getClient(): Redis { - if (!this.redisClient) { - try { - const redisUrl = this.configService.get('REDIS_URL') || process.env.REDIS_URL || 'redis://localhost:6379'; - this.redisClient = new Redis(redisUrl, { - lazyConnect: true, - enableReadyCheck: true, - maxRetriesPerRequest: 1, - }); - } catch (err) { - this.logger.warn(`Failed to initialize Redis client for health check: ${err}`); - } - } - if (!this.redisClient) { - throw new Error('Redis client could not be initialized'); - } - return this.redisClient; - } + constructor(@Inject(REDIS_CLIENT) private readonly redis: Redis) {} - async checkHealth(): Promise { + async checkHealth( + timeoutMs: number = RedisHealthIndicator.DEFAULT_TIMEOUT_MS, + ): Promise { const start = Date.now(); try { - const client = this.getClient(); - const res = await client.ping(); + // A client that has been explicitly closed will never reconnect; fail + // fast instead of waiting for the timeout. + if (this.redis.status === 'end') { + throw new Error('Redis connection is closed'); + } + + const res = await this.withTimeout(this.redis.ping(), timeoutMs); const latencyMs = Date.now() - start; if (res !== 'PONG') { @@ -54,7 +55,7 @@ export class RedisHealthIndicator { } catch (error) { const latencyMs = Date.now() - start; const message = error instanceof Error ? error.message : String(error); - this.logger.error(`Redis health check failed: ${message}`); + this.logger.error(`Redis health check failed after ${latencyMs}ms: ${message}`); return { status: 'down', @@ -64,4 +65,27 @@ export class RedisHealthIndicator { }; } } + + private withTimeout(probe: Promise, timeoutMs: number): Promise { + if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) { + return probe; + } + + return new Promise((resolve, reject) => { + const timer = setTimeout(() => { + reject(new Error(`Redis health check timed out after ${timeoutMs}ms`)); + }, timeoutMs); + + probe.then( + (value) => { + clearTimeout(timer); + resolve(value); + }, + (error: unknown) => { + clearTimeout(timer); + reject(error instanceof Error ? error : new Error(String(error))); + }, + ); + }); + } }