diff --git a/CHANGELOG.md b/CHANGELOG.md
index 92fdc34be..78bcd0988 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
## [Unreleased]
+### Added
+- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)
+
### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.
diff --git a/docs/docs.json b/docs/docs.json
index ad45e7fc8..ae26d3c8d 100644
--- a/docs/docs.json
+++ b/docs/docs.json
@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
- "GET /api/health"
+ "GET /api/health",
+ "docs/api-reference/health"
]
}
]
diff --git a/docs/docs/api-reference/health.mdx b/docs/docs/api-reference/health.mdx
new file mode 100644
index 000000000..3550dd697
--- /dev/null
+++ b/docs/docs/api-reference/health.mdx
@@ -0,0 +1,97 @@
+---
+title: "Health Endpoints"
+description: "Liveness and readiness probes for orchestrators and monitoring systems."
+---
+
+Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.
+
+## Liveness: `GET /api/health`
+
+Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.
+
+```bash
+curl -fsS https://sourcebot.example.com/api/health
+# {"status":"ok"}
+```
+
+## Readiness: `GET /api/health/ready`
+
+Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.
+
+Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.
+
+### Response shape
+
+```json
+{
+ "status": "ok",
+ "checks": {
+ "postgres": { "status": "ok", "latencyMs": 3 },
+ "redis": { "status": "ok", "latencyMs": 1 },
+ "zoekt": { "status": "ok", "latencyMs": 12 }
+ }
+}
+```
+
+When degraded, each failed check carries an `error` field with the underlying message:
+
+```json
+{
+ "status": "degraded",
+ "checks": {
+ "postgres": { "status": "ok", "latencyMs": 4 },
+ "redis": { "status": "ok", "latencyMs": 1 },
+ "zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
+ }
+}
+```
+
+| Check | What it probes |
+|-------|---------------|
+| `postgres` | `SELECT 1` via Prisma |
+| `redis` | `PING` (rejects non-`PONG` responses) |
+| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |
+
+### Example probes
+
+
+
+ ```yaml
+ services:
+ sourcebot:
+ image: sourcebot/sourcebot:latest
+ healthcheck:
+ test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
+ interval: 30s
+ timeout: 5s
+ retries: 3
+ # For dependency-aware probes, point the orchestrator at /api/health/ready
+ # instead. Sourcebot's example compose file does this via a sidecar.
+ ```
+
+
+ ```yaml
+ livenessProbe:
+ httpGet:
+ path: /api/health
+ port: 3000
+ initialDelaySeconds: 30
+ periodSeconds: 30
+ timeoutSeconds: 5
+ failureThreshold: 3
+ readinessProbe:
+ httpGet:
+ path: /api/health/ready
+ port: 3000
+ initialDelaySeconds: 10
+ periodSeconds: 10
+ timeoutSeconds: 5
+ successThreshold: 1
+ failureThreshold: 3
+ ```
+
+
+
+
+The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
+
diff --git a/packages/web/src/app/api/(server)/health/ready/route.test.ts b/packages/web/src/app/api/(server)/health/ready/route.test.ts
new file mode 100644
index 000000000..a7203af01
--- /dev/null
+++ b/packages/web/src/app/api/(server)/health/ready/route.test.ts
@@ -0,0 +1,199 @@
+import { beforeEach, describe, expect, test, vi } from 'vitest';
+
+const mocks = vi.hoisted(() => ({
+ unsafePrisma: {
+ $queryRaw: vi.fn(),
+ },
+ redisPing: vi.fn(),
+ zoektList: vi.fn(),
+}));
+
+vi.mock('server-only', () => ({}));
+
+vi.mock('@/prisma', () => ({
+ __unsafePrisma: mocks.unsafePrisma,
+}));
+
+vi.mock('@/lib/redis', () => ({
+ getRedisClient: () => ({
+ ping: mocks.redisPing,
+ }),
+}));
+
+vi.mock('@/lib/posthog', () => ({
+ captureEvent: vi.fn(),
+}));
+
+vi.mock('@/lib/zoektClient', () => ({
+ loadZoektClient: () => ({
+ List: mocks.zoektList,
+ }),
+}));
+
+vi.mock('@sourcebot/shared', () => ({
+ createLogger: () => ({
+ debug: vi.fn(),
+ info: vi.fn(),
+ warn: vi.fn(),
+ error: vi.fn(),
+ }),
+}));
+
+const { GET } = await import('./route');
+
+describe('GET /api/health/ready', () => {
+ beforeEach(() => {
+ vi.clearAllMocks();
+ mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
+ mocks.redisPing.mockResolvedValue('PONG');
+ mocks.zoektList.mockImplementation(
+ (_request: unknown, callback: (err: Error | null) => void) => {
+ callback(null);
+ },
+ );
+ });
+
+ test('returns 200 with status:ok when all three dependencies are reachable', async () => {
+ const response = await GET();
+ const body = await response.json();
+
+ expect(response.status).toBe(200);
+ expect(body.status).toBe('ok');
+ expect(body.checks.postgres.status).toBe('ok');
+ expect(body.checks.redis.status).toBe('ok');
+ expect(body.checks.zoekt.status).toBe('ok');
+ expect(typeof body.checks.postgres.latencyMs).toBe('number');
+ expect(typeof body.checks.redis.latencyMs).toBe('number');
+ expect(typeof body.checks.zoekt.latencyMs).toBe('number');
+ });
+
+ test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
+ mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));
+
+ const response = await GET();
+ const body = await response.json();
+
+ expect(response.status).toBe(503);
+ expect(body.status).toBe('degraded');
+ expect(body.checks.postgres.status).toBe('error');
+ expect(body.checks.postgres.error).toBe('connection refused');
+ expect(body.checks.redis.status).toBe('ok');
+ expect(body.checks.zoekt.status).toBe('ok');
+ });
+
+ test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
+ mocks.redisPing.mockRejectedValue(new Error('redis down'));
+
+ const response = await GET();
+ const body = await response.json();
+
+ expect(response.status).toBe(503);
+ expect(body.status).toBe('degraded');
+ expect(body.checks.postgres.status).toBe('ok');
+ expect(body.checks.redis.status).toBe('error');
+ expect(body.checks.redis.error).toBe('redis down');
+ expect(body.checks.zoekt.status).toBe('ok');
+ });
+
+ test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
+ mocks.zoektList.mockImplementation(
+ (_request: unknown, callback: (err: Error | null) => void) => {
+ callback(new Error('UNAVAILABLE: zoekt not reachable'));
+ },
+ );
+
+ const response = await GET();
+ const body = await response.json();
+
+ expect(response.status).toBe(503);
+ expect(body.status).toBe('degraded');
+ expect(body.checks.zoekt.status).toBe('error');
+ expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
+ expect(body.checks.postgres.status).toBe('ok');
+ expect(body.checks.redis.status).toBe('ok');
+ });
+
+ test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
+ mocks.redisPing.mockResolvedValue('NOT-PONG');
+
+ const response = await GET();
+ const body = await response.json();
+
+ expect(response.status).toBe(503);
+ expect(body.status).toBe('degraded');
+ expect(body.checks.redis.status).toBe('error');
+ expect(body.checks.redis.error).toContain('unexpected ping response');
+ });
+
+ test('runs all three checks in parallel (Promise.all)', async () => {
+ const delay = 50;
+ mocks.unsafePrisma.$queryRaw.mockImplementation(
+ () => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
+ );
+ mocks.redisPing.mockImplementation(
+ () => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
+ );
+ mocks.zoektList.mockImplementation(
+ (_request: unknown, callback: (err: Error | null) => void) => {
+ setTimeout(() => callback(null), delay);
+ },
+ );
+
+ const start = Date.now();
+ const response = await GET();
+ const elapsed = Date.now() - start;
+ const body = await response.json();
+
+ expect(response.status).toBe(200);
+ expect(body.status).toBe('ok');
+ // Generous upper bound to avoid flakes; serial would be ~3x delay.
+ expect(elapsed).toBeLessThan(delay * 2.5);
+ });
+
+ test('does not surface check rejections as unhandled promise rejections', async () => {
+ // The check rejects synchronously (well within the 2s timeout). The
+ // no-op `.catch` attached in `withTimeout` must absorb that
+ // rejection so the Node process does not log an
+ // unhandled-promise-rejection warning while the readiness request
+ // has already moved on.
+ const checkRejection = new Error('check rejected');
+ const unhandled: unknown[] = [];
+ const onUnhandled = (err: unknown) => { unhandled.push(err); };
+ process.on('unhandledRejection', onUnhandled);
+
+ try {
+ mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
+ mocks.redisPing.mockResolvedValue('PONG');
+ mocks.zoektList.mockImplementation(
+ (_request: unknown, callback: (err: Error | null) => void) => {
+ callback(null);
+ },
+ );
+
+ const response = await GET();
+ const body = await response.json();
+
+ expect(response.status).toBe(503);
+ expect(body.checks.postgres.status).toBe('error');
+ // Give the rejection microtask a chance to fire and propagate.
+ await new Promise((resolve) => setTimeout(resolve, 50));
+ expect(unhandled).not.toContain(checkRejection);
+ } finally {
+ process.off('unhandledRejection', onUnhandled);
+ }
+ });
+
+ test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
+ // Regression guard: the earlier draft of the Zoekt probe passed
+ // `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
+ // belongs to `SearchOptions` and is silently ignored by `List`
+ // (whose `ListOptions` only carries `field`). The 2s client-side
+ // timeout is the only thing that actually bounds the call. The
+ // probe must therefore issue the smallest valid request, which is
+ // an empty options object.
+ const response = await GET();
+ expect(response.status).toBe(200);
+ expect(mocks.zoektList).toHaveBeenCalledTimes(1);
+ expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
+ });
+});
diff --git a/packages/web/src/app/api/(server)/health/ready/route.ts b/packages/web/src/app/api/(server)/health/ready/route.ts
new file mode 100644
index 000000000..3bb072e13
--- /dev/null
+++ b/packages/web/src/app/api/(server)/health/ready/route.ts
@@ -0,0 +1,147 @@
+import { createLogger } from '@sourcebot/shared';
+
+import { apiHandler } from '@/lib/apiHandler';
+import { __unsafePrisma } from '@/prisma';
+import { getRedisClient } from '@/lib/redis';
+import { loadZoektClient } from '@/lib/zoektClient';
+
+// Per-check timeout. The three checks run in parallel, so the worst-case
+// request time is bounded by this value even when one dependency hangs.
+const READINESS_TIMEOUT_MS = 2000;
+
+const logger = createLogger('health-ready');
+
+type CheckStatus = 'ok' | 'error';
+type CheckResult = {
+ status: CheckStatus;
+ latencyMs: number;
+ error?: string;
+};
+type ReadinessResponse = {
+ status: 'ok' | 'degraded';
+ checks: {
+ postgres: CheckResult;
+ redis: CheckResult;
+ zoekt: CheckResult;
+ };
+};
+
+// Runs `check()` and rejects if it has not settled after `timeoutMs`. When the
+// timeout fires, the underlying check promise may still resolve or reject
+// later; the no-op `.catch` below attaches to that promise so a late
+// rejection does not surface as an unhandled-promise-rejection in the Node
+// process while the readiness request has already moved on.
+const withTimeout = async (
+ label: string,
+ check: () => Promise,
+ timeoutMs: number,
+): Promise => {
+ const checkPromise = check();
+ checkPromise.catch(() => { /* swallowed: see comment above */ });
+
+ let timer: ReturnType | undefined;
+ const timeout = new Promise((_, reject) => {
+ timer = setTimeout(() => {
+ reject(new Error(`${label} check timed out after ${timeoutMs}ms`));
+ }, timeoutMs);
+ });
+ try {
+ return await Promise.race([checkPromise, timeout]);
+ } finally {
+ if (timer) {
+ clearTimeout(timer);
+ }
+ }
+};
+
+const checkPostgres = async (): Promise => {
+ const start = Date.now();
+ try {
+ await withTimeout('postgres', async () => {
+ await __unsafePrisma.$queryRaw`SELECT 1`;
+ }, READINESS_TIMEOUT_MS);
+ return { status: 'ok', latencyMs: Date.now() - start };
+ } catch (err) {
+ return {
+ status: 'error',
+ latencyMs: Date.now() - start,
+ error: err instanceof Error ? err.message : String(err),
+ };
+ }
+};
+
+const checkRedis = async (): Promise => {
+ const start = Date.now();
+ try {
+ const redis = getRedisClient();
+ await withTimeout('redis', async () => {
+ const pong = await redis.ping();
+ if (pong !== 'PONG') {
+ throw new Error(`unexpected ping response: ${pong}`);
+ }
+ }, READINESS_TIMEOUT_MS);
+ return { status: 'ok', latencyMs: Date.now() - start };
+ } catch (err) {
+ return {
+ status: 'error',
+ latencyMs: Date.now() - start,
+ error: err instanceof Error ? err.message : String(err),
+ };
+ }
+};
+
+const checkZoekt = async (): Promise => {
+ const start = Date.now();
+ try {
+ await withTimeout('zoekt', async () => {
+ // Build the client inside the timeout so a first-call init stall
+ // (e.g., vendored proto load) is also bounded.
+ const client = await loadZoektClient();
+ await new Promise((resolve, reject) => {
+ // Empty `List` is the smallest request that exercises the gRPC
+ // channel end-to-end. It returns an empty result, not an error,
+ // even when no repos are indexed. The 2s client-side
+ // `READINESS_TIMEOUT_MS` is the only timeout that actually
+ // bounds the call: `max_wall_time` is a `SearchOptions` field
+ // and does not apply to `List` (`ListOptions` only carries
+ // `field`).
+ client.List({}, (err) => {
+ if (err) {
+ reject(err);
+ } else {
+ resolve();
+ }
+ });
+ });
+ }, READINESS_TIMEOUT_MS);
+ return { status: 'ok', latencyMs: Date.now() - start };
+ } catch (err) {
+ return {
+ status: 'error',
+ latencyMs: Date.now() - start,
+ error: err instanceof Error ? err.message : String(err),
+ };
+ }
+};
+
+// eslint-disable-next-line authz/require-auth-wrapper -- public readiness probe, no user data returned
+export const GET = apiHandler(async () => {
+ const [postgres, redis, zoekt] = await Promise.all([
+ checkPostgres(),
+ checkRedis(),
+ checkZoekt(),
+ ]);
+
+ const checks = { postgres, redis, zoekt };
+ const healthy = postgres.status === 'ok' && redis.status === 'ok' && zoekt.status === 'ok';
+
+ if (!healthy) {
+ logger.warn('readiness check failed', { checks });
+ }
+
+ const body: ReadinessResponse = {
+ status: healthy ? 'ok' : 'degraded',
+ checks,
+ };
+ return Response.json(body, { status: healthy ? 200 : 503 });
+}, { track: false });
diff --git a/packages/web/src/lib/zoektClient.ts b/packages/web/src/lib/zoektClient.ts
new file mode 100644
index 000000000..accaf3963
--- /dev/null
+++ b/packages/web/src/lib/zoektClient.ts
@@ -0,0 +1,58 @@
+// Thin wrapper around the vendored Zoekt webserver gRPC client. Kept in
+// a separate module so the readiness route can mock it in tests without
+// pulling the gRPC + @opentelemetry CJS chain at test-load time.
+
+export type ZoektListRequest = {
+ opts?: { max_wall_time?: { seconds: number; nanos: number } };
+};
+
+export type ZoektClient = {
+ List: (request: ZoektListRequest, callback: (err: Error | null) => void) => void;
+};
+
+let cachedClient: ZoektClient | undefined;
+
+const buildClient = async (): Promise => {
+ const [grpc, protoLoader, nodePath, shared] = await Promise.all([
+ import('@grpc/grpc-js'),
+ import('@grpc/proto-loader'),
+ import('node:path'),
+ import('@sourcebot/shared'),
+ ]);
+
+ const protoBasePath = nodePath.join(process.cwd(), '../../vendor/zoekt/grpc/protos');
+ const protoPath = nodePath.join(protoBasePath, 'zoekt/webserver/v1/webserver.proto');
+
+ const packageDefinition = protoLoader.loadSync(protoPath, {
+ keepCase: true,
+ longs: Number,
+ enums: String,
+ defaults: true,
+ oneofs: true,
+ includeDirs: [protoBasePath],
+ });
+
+ const proto = grpc.loadPackageDefinition(packageDefinition) as unknown as {
+ zoekt: { webserver: { v1: { WebserverService: new (address: string, credentials: unknown) => ZoektClient } } };
+ };
+
+ const zoektUrl = new URL(shared.env.ZOEKT_WEBSERVER_URL);
+ const grpcAddress = `${zoektUrl.hostname}:${zoektUrl.port}`;
+ return new proto.zoekt.webserver.v1.WebserverService(
+ grpcAddress,
+ grpc.credentials.createInsecure(),
+ );
+};
+
+// Returns a connected client, building it on first use. Only successful
+// builds are cached: a transient init failure does not poison subsequent
+// readiness probes, so the next call can retry.
+export const loadZoektClient = async (): Promise => {
+ if (cachedClient) {
+ return cachedClient;
+ }
+ const client = await buildClient();
+ cachedClient = client;
+ return client;
+};
+