Skip to content
Closed
Show file tree
Hide file tree
Changes from 9 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line number Diff line number Diff line change
Expand Up @@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | `List` RPC with a 1-second wall-time cap (proves the gRPC channel is alive) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
185 changes: 185 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,185 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});
});
Loading