Health checks enable monitoring systems, load balancers, and orchestrators to assess application health and make routing decisions. This practice defines how to implement comprehensive health checking.
Health Check Types
Health Check Endpoints
Standard Endpoints
| Endpoint | Purpose | HTTP Status |
|---|---|---|
/health | Overall health | 200 OK / 503 Service Unavailable |
/health/live | Liveness check | 200 OK / 503 Service Unavailable |
/health/ready | Readiness check | 200 OK / 503 Service Unavailable |
/health/startup | Startup check | 200 OK / 503 Service Unavailable |
Response Format
{
"status": "healthy",
"version": "1.2.3",
"timestamp": "2024-01-15T10:30:00Z",
"checks": {
"database": {
"status": "healthy",
"latency_ms": 5
},
"cache": {
"status": "healthy",
"latency_ms": 1
},
"external_api": {
"status": "healthy",
"latency_ms": 45
}
}
}
Unhealthy Response
{
"status": "unhealthy",
"version": "1.2.3",
"timestamp": "2024-01-15T10:30:00Z",
"checks": {
"database": {
"status": "unhealthy",
"error": "connection timeout",
"latency_ms": 5000
},
"cache": {
"status": "healthy",
"latency_ms": 1
}
}
}
Implementation Examples
Node.js / Express
import express from 'express';
import { Pool } from 'pg';
import Redis from 'ioredis';
const app = express();
const db = new Pool({ connectionString: process.env.DATABASE_URL });
const redis = new Redis(process.env.REDIS_URL);
interface HealthCheck {
status: 'healthy' | 'unhealthy';
latency_ms?: number;
error?: string;
}
interface HealthResponse {
status: 'healthy' | 'unhealthy';
version: string;
timestamp: string;
checks: Record<string, HealthCheck>;
}
async function checkDatabase(): Promise<HealthCheck> {
const start = Date.now();
try {
await db.query('SELECT 1');
return { status: 'healthy', latency_ms: Date.now() - start };
} catch (error) {
return {
status: 'unhealthy',
latency_ms: Date.now() - start,
error: error.message,
};
}
}
async function checkRedis(): Promise<HealthCheck> {
const start = Date.now();
try {
await redis.ping();
return { status: 'healthy', latency_ms: Date.now() - start };
} catch (error) {
return {
status: 'unhealthy',
latency_ms: Date.now() - start,
error: error.message,
};
}
}
// Liveness - just check if process is running
app.get('/health/live', (req, res) => {
res.json({ status: 'healthy' });
});
// Readiness - check all dependencies
app.get('/health/ready', async (req, res) => {
const checks = {
database: await checkDatabase(),
cache: await checkRedis(),
};
const isHealthy = Object.values(checks).every(c => c.status === 'healthy');
const response: HealthResponse = {
status: isHealthy ? 'healthy' : 'unhealthy',
version: process.env.APP_VERSION || '0.0.0',
timestamp: new Date().toISOString(),
checks,
};
res.status(isHealthy ? 200 : 503).json(response);
});
// Full health check
app.get('/health', async (req, res) => {
const checks = {
database: await checkDatabase(),
cache: await checkRedis(),
};
const isHealthy = Object.values(checks).every(c => c.status === 'healthy');
const response: HealthResponse = {
status: isHealthy ? 'healthy' : 'unhealthy',
version: process.env.APP_VERSION || '0.0.0',
timestamp: new Date().toISOString(),
checks,
};
res.status(isHealthy ? 200 : 503).json(response);
});
Python / FastAPI
from fastapi import FastAPI, Response
from datetime import datetime
import asyncpg
import aioredis
import os
from typing import Optional
from pydantic import BaseModel
app = FastAPI()
class HealthCheck(BaseModel):
status: str
latency_ms: Optional[int] = None
error: Optional[str] = None
class HealthResponse(BaseModel):
status: str
version: str
timestamp: str
checks: dict[str, HealthCheck]
async def check_database() -> HealthCheck:
start = datetime.now()
try:
conn = await asyncpg.connect(os.environ["DATABASE_URL"])
await conn.execute("SELECT 1")
await conn.close()
latency = (datetime.now() - start).total_seconds() * 1000
return HealthCheck(status="healthy", latency_ms=int(latency))
except Exception as e:
latency = (datetime.now() - start).total_seconds() * 1000
return HealthCheck(status="unhealthy", latency_ms=int(latency), error=str(e))
async def check_redis() -> HealthCheck:
start = datetime.now()
try:
redis = aioredis.from_url(os.environ["REDIS_URL"])
await redis.ping()
await redis.close()
latency = (datetime.now() - start).total_seconds() * 1000
return HealthCheck(status="healthy", latency_ms=int(latency))
except Exception as e:
latency = (datetime.now() - start).total_seconds() * 1000
return HealthCheck(status="unhealthy", latency_ms=int(latency), error=str(e))
@app.get("/health/live")
async def liveness():
return {"status": "healthy"}
@app.get("/health/ready")
async def readiness(response: Response):
checks = {
"database": await check_database(),
"cache": await check_redis(),
}
is_healthy = all(c.status == "healthy" for c in checks.values())
health_response = HealthResponse(
status="healthy" if is_healthy else "unhealthy",
version=os.environ.get("APP_VERSION", "0.0.0"),
timestamp=datetime.utcnow().isoformat() + "Z",
checks={k: v.dict() for k, v in checks.items()},
)
if not is_healthy:
response.status_code = 503
return health_response
@app.get("/health")
async def health(response: Response):
return await readiness(response)
.NET / ASP.NET Core
// Program.cs
using Microsoft.Extensions.Diagnostics.HealthChecks;
var builder = WebApplication.CreateBuilder(args);
builder.Services.AddHealthChecks()
.AddNpgSql(builder.Configuration.GetConnectionString("Database")!, name: "database")
.AddRedis(builder.Configuration.GetConnectionString("Redis")!, name: "cache")
.AddUrlGroup(new Uri("https://api.external.com/health"), name: "external-api");
var app = builder.Build();
app.MapHealthChecks("/health/live", new HealthCheckOptions
{
Predicate = _ => false, // No checks, just process alive
ResponseWriter = WriteResponse
});
app.MapHealthChecks("/health/ready", new HealthCheckOptions
{
Predicate = check => check.Tags.Contains("ready"),
ResponseWriter = WriteResponse
});
app.MapHealthChecks("/health", new HealthCheckOptions
{
ResponseWriter = WriteResponse
});
static Task WriteResponse(HttpContext context, HealthReport report)
{
context.Response.ContentType = "application/json";
var response = new
{
status = report.Status.ToString().ToLower(),
version = Environment.GetEnvironmentVariable("APP_VERSION") ?? "0.0.0",
timestamp = DateTime.UtcNow.ToString("O"),
checks = report.Entries.ToDictionary(
e => e.Key,
e => new
{
status = e.Value.Status.ToString().ToLower(),
latency_ms = (int)e.Value.Duration.TotalMilliseconds,
error = e.Value.Exception?.Message
}
)
};
return context.Response.WriteAsJsonAsync(response);
}
Kubernetes Configuration
Health Check Probes
apiVersion: apps/v1
kind: Deployment
metadata:
name: my-app
spec:
template:
spec:
containers:
- name: my-app
image: my-app:1.0.0
ports:
- containerPort: 8080
# Liveness probe - restart if fails
livenessProbe:
httpGet:
path: /health/live
port: 8080
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 3
# Readiness probe - remove from service if fails
readinessProbe:
httpGet:
path: /health/ready
port: 8080
initialDelaySeconds: 5
periodSeconds: 5
timeoutSeconds: 3
failureThreshold: 3
# Startup probe - for slow-starting apps
startupProbe:
httpGet:
path: /health/startup
port: 8080
initialDelaySeconds: 0
periodSeconds: 5
timeoutSeconds: 3
failureThreshold: 30 # 30 * 5s = 150s max startup time
Probe Tuning Guidelines
| Parameter | Liveness | Readiness | Startup |
|---|---|---|---|
| initialDelaySeconds | App startup time | 0-5s | 0 |
| periodSeconds | 10-30s | 5-10s | 5-10s |
| timeoutSeconds | 5s | 3s | 3s |
| failureThreshold | 3 | 3 | Based on max startup |
Load Balancer Health Checks
AWS ALB
resource "aws_lb_target_group" "app" {
name = "app-tg"
port = 8080
protocol = "HTTP"
vpc_id = aws_vpc.main.id
health_check {
enabled = true
healthy_threshold = 2
unhealthy_threshold = 3
timeout = 5
interval = 30
path = "/health/ready"
port = "traffic-port"
protocol = "HTTP"
matcher = "200"
}
}
nginx
upstream backend {
server app1:8080;
server app2:8080;
# Health check (nginx plus or third-party module)
health_check interval=10s
fails=3
passes=2
uri=/health/ready;
}
Monitoring Integration
Prometheus Metrics
import client from 'prom-client';
const healthCheckGauge = new client.Gauge({
name: 'app_health_check_status',
help: 'Health check status (1 = healthy, 0 = unhealthy)',
labelNames: ['check'],
});
const healthCheckLatency = new client.Histogram({
name: 'app_health_check_latency_seconds',
help: 'Health check latency in seconds',
labelNames: ['check'],
buckets: [0.001, 0.005, 0.01, 0.05, 0.1, 0.5, 1],
});
// Update metrics after health check
function recordHealthCheck(name: string, status: boolean, latencyMs: number) {
healthCheckGauge.set({ check: name }, status ? 1 : 0);
healthCheckLatency.observe({ check: name }, latencyMs / 1000);
}
Best Practices
Do's
- Keep liveness checks simple and fast
- Include all critical dependencies in readiness checks
- Use appropriate timeouts (not too short, not too long)
- Return structured JSON responses
- Include version information in responses
- Log health check failures for debugging
Don'ts
- Don't include non-critical dependencies in liveness checks
- Don't make liveness checks depend on external services
- Don't use the same probe for liveness and readiness
- Don't set timeouts longer than the check interval
- Don't include sensitive information in responses
Related Resources
- Observability - Monitoring and logging
- Infrastructure Tools - Deployment configuration
- CI/CD & DevOps - Deployment pipelines
Compliance
This section fulfills ISO 13485 requirements for monitoring and measurement (8.2.4) and infrastructure maintenance (6.3), and ISO 27001 requirements for monitoring activities (A.8.16), event logging (A.8.15), and availability (A.8.14).