Health Checks

Practices

Application health monitoring practices for NUP projects

Health checks enable monitoring systems, load balancers, and orchestrators to assess application health and make routing decisions. This practice defines how to implement comprehensive health checking.

Health Check Types

Health Check Hierarchy
Health Check Hierarchy

Health Check Endpoints

Standard Endpoints

EndpointPurposeHTTP Status
/healthOverall health200 OK / 503 Service Unavailable
/health/liveLiveness check200 OK / 503 Service Unavailable
/health/readyReadiness check200 OK / 503 Service Unavailable
/health/startupStartup check200 OK / 503 Service Unavailable

Response Format

{
  "status": "healthy",
  "version": "1.2.3",
  "timestamp": "2024-01-15T10:30:00Z",
  "checks": {
    "database": {
      "status": "healthy",
      "latency_ms": 5
    },
    "cache": {
      "status": "healthy",
      "latency_ms": 1
    },
    "external_api": {
      "status": "healthy",
      "latency_ms": 45
    }
  }
}

Unhealthy Response

{
  "status": "unhealthy",
  "version": "1.2.3",
  "timestamp": "2024-01-15T10:30:00Z",
  "checks": {
    "database": {
      "status": "unhealthy",
      "error": "connection timeout",
      "latency_ms": 5000
    },
    "cache": {
      "status": "healthy",
      "latency_ms": 1
    }
  }
}

Implementation Examples

Node.js / Express

import express from 'express';
import { Pool } from 'pg';
import Redis from 'ioredis';

const app = express();
const db = new Pool({ connectionString: process.env.DATABASE_URL });
const redis = new Redis(process.env.REDIS_URL);

interface HealthCheck {
  status: 'healthy' | 'unhealthy';
  latency_ms?: number;
  error?: string;
}

interface HealthResponse {
  status: 'healthy' | 'unhealthy';
  version: string;
  timestamp: string;
  checks: Record<string, HealthCheck>;
}

async function checkDatabase(): Promise<HealthCheck> {
  const start = Date.now();
  try {
    await db.query('SELECT 1');
    return { status: 'healthy', latency_ms: Date.now() - start };
  } catch (error) {
    return {
      status: 'unhealthy',
      latency_ms: Date.now() - start,
      error: error.message,
    };
  }
}

async function checkRedis(): Promise<HealthCheck> {
  const start = Date.now();
  try {
    await redis.ping();
    return { status: 'healthy', latency_ms: Date.now() - start };
  } catch (error) {
    return {
      status: 'unhealthy',
      latency_ms: Date.now() - start,
      error: error.message,
    };
  }
}

// Liveness - just check if process is running
app.get('/health/live', (req, res) => {
  res.json({ status: 'healthy' });
});

// Readiness - check all dependencies
app.get('/health/ready', async (req, res) => {
  const checks = {
    database: await checkDatabase(),
    cache: await checkRedis(),
  };

  const isHealthy = Object.values(checks).every(c => c.status === 'healthy');

  const response: HealthResponse = {
    status: isHealthy ? 'healthy' : 'unhealthy',
    version: process.env.APP_VERSION || '0.0.0',
    timestamp: new Date().toISOString(),
    checks,
  };

  res.status(isHealthy ? 200 : 503).json(response);
});

// Full health check
app.get('/health', async (req, res) => {
  const checks = {
    database: await checkDatabase(),
    cache: await checkRedis(),
  };

  const isHealthy = Object.values(checks).every(c => c.status === 'healthy');

  const response: HealthResponse = {
    status: isHealthy ? 'healthy' : 'unhealthy',
    version: process.env.APP_VERSION || '0.0.0',
    timestamp: new Date().toISOString(),
    checks,
  };

  res.status(isHealthy ? 200 : 503).json(response);
});

Python / FastAPI

from fastapi import FastAPI, Response
from datetime import datetime
import asyncpg
import aioredis
import os
from typing import Optional
from pydantic import BaseModel

app = FastAPI()

class HealthCheck(BaseModel):
    status: str
    latency_ms: Optional[int] = None
    error: Optional[str] = None

class HealthResponse(BaseModel):
    status: str
    version: str
    timestamp: str
    checks: dict[str, HealthCheck]

async def check_database() -> HealthCheck:
    start = datetime.now()
    try:
        conn = await asyncpg.connect(os.environ["DATABASE_URL"])
        await conn.execute("SELECT 1")
        await conn.close()
        latency = (datetime.now() - start).total_seconds() * 1000
        return HealthCheck(status="healthy", latency_ms=int(latency))
    except Exception as e:
        latency = (datetime.now() - start).total_seconds() * 1000
        return HealthCheck(status="unhealthy", latency_ms=int(latency), error=str(e))

async def check_redis() -> HealthCheck:
    start = datetime.now()
    try:
        redis = aioredis.from_url(os.environ["REDIS_URL"])
        await redis.ping()
        await redis.close()
        latency = (datetime.now() - start).total_seconds() * 1000
        return HealthCheck(status="healthy", latency_ms=int(latency))
    except Exception as e:
        latency = (datetime.now() - start).total_seconds() * 1000
        return HealthCheck(status="unhealthy", latency_ms=int(latency), error=str(e))

@app.get("/health/live")
async def liveness():
    return {"status": "healthy"}

@app.get("/health/ready")
async def readiness(response: Response):
    checks = {
        "database": await check_database(),
        "cache": await check_redis(),
    }

    is_healthy = all(c.status == "healthy" for c in checks.values())

    health_response = HealthResponse(
        status="healthy" if is_healthy else "unhealthy",
        version=os.environ.get("APP_VERSION", "0.0.0"),
        timestamp=datetime.utcnow().isoformat() + "Z",
        checks={k: v.dict() for k, v in checks.items()},
    )

    if not is_healthy:
        response.status_code = 503

    return health_response

@app.get("/health")
async def health(response: Response):
    return await readiness(response)

.NET / ASP.NET Core

// Program.cs
using Microsoft.Extensions.Diagnostics.HealthChecks;

var builder = WebApplication.CreateBuilder(args);

builder.Services.AddHealthChecks()
    .AddNpgSql(builder.Configuration.GetConnectionString("Database")!, name: "database")
    .AddRedis(builder.Configuration.GetConnectionString("Redis")!, name: "cache")
    .AddUrlGroup(new Uri("https://api.external.com/health"), name: "external-api");

var app = builder.Build();

app.MapHealthChecks("/health/live", new HealthCheckOptions
{
    Predicate = _ => false, // No checks, just process alive
    ResponseWriter = WriteResponse
});

app.MapHealthChecks("/health/ready", new HealthCheckOptions
{
    Predicate = check => check.Tags.Contains("ready"),
    ResponseWriter = WriteResponse
});

app.MapHealthChecks("/health", new HealthCheckOptions
{
    ResponseWriter = WriteResponse
});

static Task WriteResponse(HttpContext context, HealthReport report)
{
    context.Response.ContentType = "application/json";

    var response = new
    {
        status = report.Status.ToString().ToLower(),
        version = Environment.GetEnvironmentVariable("APP_VERSION") ?? "0.0.0",
        timestamp = DateTime.UtcNow.ToString("O"),
        checks = report.Entries.ToDictionary(
            e => e.Key,
            e => new
            {
                status = e.Value.Status.ToString().ToLower(),
                latency_ms = (int)e.Value.Duration.TotalMilliseconds,
                error = e.Value.Exception?.Message
            }
        )
    };

    return context.Response.WriteAsJsonAsync(response);
}

Kubernetes Configuration

Health Check Probes

apiVersion: apps/v1
kind: Deployment
metadata:
  name: my-app
spec:
  template:
    spec:
      containers:
        - name: my-app
          image: my-app:1.0.0
          ports:
            - containerPort: 8080

          # Liveness probe - restart if fails
          livenessProbe:
            httpGet:
              path: /health/live
              port: 8080
            initialDelaySeconds: 10
            periodSeconds: 10
            timeoutSeconds: 5
            failureThreshold: 3

          # Readiness probe - remove from service if fails
          readinessProbe:
            httpGet:
              path: /health/ready
              port: 8080
            initialDelaySeconds: 5
            periodSeconds: 5
            timeoutSeconds: 3
            failureThreshold: 3

          # Startup probe - for slow-starting apps
          startupProbe:
            httpGet:
              path: /health/startup
              port: 8080
            initialDelaySeconds: 0
            periodSeconds: 5
            timeoutSeconds: 3
            failureThreshold: 30  # 30 * 5s = 150s max startup time

Probe Tuning Guidelines

ParameterLivenessReadinessStartup
initialDelaySecondsApp startup time0-5s0
periodSeconds10-30s5-10s5-10s
timeoutSeconds5s3s3s
failureThreshold33Based on max startup

Load Balancer Health Checks

AWS ALB

resource "aws_lb_target_group" "app" {
  name     = "app-tg"
  port     = 8080
  protocol = "HTTP"
  vpc_id   = aws_vpc.main.id

  health_check {
    enabled             = true
    healthy_threshold   = 2
    unhealthy_threshold = 3
    timeout             = 5
    interval            = 30
    path                = "/health/ready"
    port                = "traffic-port"
    protocol            = "HTTP"
    matcher             = "200"
  }
}

nginx

upstream backend {
    server app1:8080;
    server app2:8080;

    # Health check (nginx plus or third-party module)
    health_check interval=10s
                 fails=3
                 passes=2
                 uri=/health/ready;
}

Monitoring Integration

Prometheus Metrics

import client from 'prom-client';

const healthCheckGauge = new client.Gauge({
  name: 'app_health_check_status',
  help: 'Health check status (1 = healthy, 0 = unhealthy)',
  labelNames: ['check'],
});

const healthCheckLatency = new client.Histogram({
  name: 'app_health_check_latency_seconds',
  help: 'Health check latency in seconds',
  labelNames: ['check'],
  buckets: [0.001, 0.005, 0.01, 0.05, 0.1, 0.5, 1],
});

// Update metrics after health check
function recordHealthCheck(name: string, status: boolean, latencyMs: number) {
  healthCheckGauge.set({ check: name }, status ? 1 : 0);
  healthCheckLatency.observe({ check: name }, latencyMs / 1000);
}

Best Practices

Do's

  • Keep liveness checks simple and fast
  • Include all critical dependencies in readiness checks
  • Use appropriate timeouts (not too short, not too long)
  • Return structured JSON responses
  • Include version information in responses
  • Log health check failures for debugging

Don'ts

  • Don't include non-critical dependencies in liveness checks
  • Don't make liveness checks depend on external services
  • Don't use the same probe for liveness and readiness
  • Don't set timeouts longer than the check interval
  • Don't include sensitive information in responses

Compliance

This section fulfills ISO 13485 requirements for monitoring and measurement (8.2.4) and infrastructure maintenance (6.3), and ISO 27001 requirements for monitoring activities (A.8.16), event logging (A.8.15), and availability (A.8.14).

View full compliance matrix

Sign in or sign up

Enter your work email to receive a temporary sign-in link.

By continuing, you agree to our Terms of Service and Privacy Policy.