From 8668e52cf1a5eac0619762aa8492f1ba0bb96eaf Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Fri, 24 Jul 2026 23:09:50 +0530 Subject: [PATCH 01/10] feat(web): extract lazy Zoekt gRPC client into a shared module The vendored Zoekt webserver gRPC client construction needs node:path, @grpc/grpc-js, @grpc/proto-loader, and the ZOEKT_WEBSERVER_URL env. Pulling those at module load time also drags the @opentelemetry CJS chain into anything that imports the client. Move the construction behind a single loadZoektClient() helper that dynamically imports the heavy deps, caches the resulting client, and lives in its own module so readiness / search / stream-search callers can import it without that boot-time cost and can mock it cleanly in tests. --- packages/web/src/lib/zoektClient.ts | 50 +++++++++++++++++++++++++++++ 1 file changed, 50 insertions(+) create mode 100644 packages/web/src/lib/zoektClient.ts diff --git a/packages/web/src/lib/zoektClient.ts b/packages/web/src/lib/zoektClient.ts new file mode 100644 index 000000000..1b02a6567 --- /dev/null +++ b/packages/web/src/lib/zoektClient.ts @@ -0,0 +1,50 @@ +// Thin wrapper around the vendored Zoekt webserver gRPC client. Kept in +// a separate module so the readiness route can mock it in tests without +// pulling the gRPC + @opentelemetry CJS chain at test-load time. + +export type ZoektListRequest = { + opts?: { max_wall_time?: { seconds: number; nanos: number } }; +}; + +export type ZoektClient = { + List: (request: ZoektListRequest, callback: (err: Error | null) => void) => void; +}; + +let zoektClientPromise: Promise | undefined; + +export const loadZoektClient = (): Promise => { + if (!zoektClientPromise) { + zoektClientPromise = (async () => { + const [grpc, protoLoader, nodePath, shared] = await Promise.all([ + import('@grpc/grpc-js'), + import('@grpc/proto-loader'), + import('node:path'), + import('@sourcebot/shared'), + ]); + + const protoBasePath = nodePath.join(process.cwd(), '../../vendor/zoekt/grpc/protos'); + const protoPath = nodePath.join(protoBasePath, 'zoekt/webserver/v1/webserver.proto'); + + const packageDefinition = protoLoader.loadSync(protoPath, { + keepCase: true, + longs: Number, + enums: String, + defaults: true, + oneofs: true, + includeDirs: [protoBasePath], + }); + + const proto = grpc.loadPackageDefinition(packageDefinition) as unknown as { + zoekt: { webserver: { v1: { WebserverService: new (address: string, credentials: unknown) => ZoektClient } } }; + }; + + const zoektUrl = new URL(shared.env.ZOEKT_WEBSERVER_URL); + const grpcAddress = `${zoektUrl.hostname}:${zoektUrl.port}`; + return new proto.zoekt.webserver.v1.WebserverService( + grpcAddress, + grpc.credentials.createInsecure(), + ); + })(); + } + return zoektClientPromise; +}; From 8e90c6fe4c42a972c5b384385ebe6dcb47baab47 Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Fri, 24 Jul 2026 23:09:57 +0530 Subject: [PATCH 02/10] feat(web): add /api/health/ready endpoint with dependency health checks Standard Kubernetes-style readiness probe. The existing /api/health liveness endpoint is left untouched. GET /api/health/ready runs three checks in parallel: - postgres: prisma.$queryRaw SELECT 1 - redis: getRedisClient().ping() (rejects non-PONG responses) - zoekt: an empty gRPC List with a 1s wall-time cap Each check is bounded by a 2s timeout, so the worst-case request time is bounded even when one dependency hangs. Returns 200 with {status:ok, checks:{...}} when all three pass, 503 with {status:degraded, checks:{...}} otherwise. Per-check payload includes status, latencyMs, and an error message when degraded. The endpoint is unauthenticated (matches the existing /api/health policy) and PostHog tracking is disabled. --- .../app/api/(server)/health/ready/route.ts | 139 ++++++++++++++++++ 1 file changed, 139 insertions(+) create mode 100644 packages/web/src/app/api/(server)/health/ready/route.ts diff --git a/packages/web/src/app/api/(server)/health/ready/route.ts b/packages/web/src/app/api/(server)/health/ready/route.ts new file mode 100644 index 000000000..2cadcbe3a --- /dev/null +++ b/packages/web/src/app/api/(server)/health/ready/route.ts @@ -0,0 +1,139 @@ +import { createLogger } from '@sourcebot/shared'; + +import { apiHandler } from '@/lib/apiHandler'; +import { __unsafePrisma } from '@/prisma'; +import { getRedisClient } from '@/lib/redis'; +import { loadZoektClient } from '@/lib/zoektClient'; + +// Per-check timeout. The three checks run in parallel, so the worst-case +// request time is bounded by this value even when one dependency hangs. +const READINESS_TIMEOUT_MS = 2000; + +const logger = createLogger('health-ready'); + +type CheckStatus = 'ok' | 'error'; +type CheckResult = { + status: CheckStatus; + latencyMs: number; + error?: string; +}; +type ReadinessResponse = { + status: 'ok' | 'degraded'; + checks: { + postgres: CheckResult; + redis: CheckResult; + zoekt: CheckResult; + }; +}; + +// Wraps a check function in a per-check timeout. When the timeout fires +// first, the check resolves as an error result; the underlying promise is +// allowed to settle in the background (its result is discarded). +const withTimeout = async ( + label: string, + check: () => Promise, + timeoutMs: number, +): Promise => { + let timer: ReturnType | undefined; + const timeout = new Promise((_, reject) => { + timer = setTimeout(() => { + reject(new Error(`${label} check timed out after ${timeoutMs}ms`)); + }, timeoutMs); + }); + try { + return await Promise.race([check(), timeout]); + } finally { + if (timer) { + clearTimeout(timer); + } + } +}; + +const checkPostgres = async (): Promise => { + const start = Date.now(); + try { + await withTimeout('postgres', async () => { + await __unsafePrisma.$queryRaw`SELECT 1`; + }, READINESS_TIMEOUT_MS); + return { status: 'ok', latencyMs: Date.now() - start }; + } catch (err) { + return { + status: 'error', + latencyMs: Date.now() - start, + error: err instanceof Error ? err.message : String(err), + }; + } +}; + +const checkRedis = async (): Promise => { + const start = Date.now(); + try { + const redis = getRedisClient(); + await withTimeout('redis', async () => { + const pong = await redis.ping(); + if (pong !== 'PONG') { + throw new Error(`unexpected ping response: ${pong}`); + } + }, READINESS_TIMEOUT_MS); + return { status: 'ok', latencyMs: Date.now() - start }; + } catch (err) { + return { + status: 'error', + latencyMs: Date.now() - start, + error: err instanceof Error ? err.message : String(err), + }; + } +}; + +const checkZoekt = async (): Promise => { + const start = Date.now(); + try { + const client = await loadZoektClient(); + await withTimeout('zoekt', async () => { + await new Promise((resolve, reject) => { + // An empty List with a 1s wall-time cap is the smallest request + // that exercises the gRPC channel end-to-end. It returns an + // empty result, not an error, even when no repos are indexed. + client.List( + { opts: { max_wall_time: { seconds: 1, nanos: 0 } } }, + (err) => { + if (err) { + reject(err); + } else { + resolve(); + } + }, + ); + }); + }, READINESS_TIMEOUT_MS); + return { status: 'ok', latencyMs: Date.now() - start }; + } catch (err) { + return { + status: 'error', + latencyMs: Date.now() - start, + error: err instanceof Error ? err.message : String(err), + }; + } +}; + +// eslint-disable-next-line authz/require-auth-wrapper -- public readiness probe, no user data returned +export const GET = apiHandler(async () => { + const [postgres, redis, zoekt] = await Promise.all([ + checkPostgres(), + checkRedis(), + checkZoekt(), + ]); + + const checks = { postgres, redis, zoekt }; + const healthy = postgres.status === 'ok' && redis.status === 'ok' && zoekt.status === 'ok'; + + if (!healthy) { + logger.warn('readiness check failed', { checks }); + } + + const body: ReadinessResponse = { + status: healthy ? 'ok' : 'degraded', + checks, + }; + return Response.json(body, { status: healthy ? 200 : 503 }); +}, { track: false }); From 0f04b34326f41c7072e4ef8cec70d93efd878ad9 Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Fri, 24 Jul 2026 23:10:05 +0530 Subject: [PATCH 03/10] test(web): cover /api/health/ready healthy and degraded paths Six cases: - 200 + status:ok when all three dependencies are reachable - 503 + postgres error when Postgres is unreachable - 503 + redis error when Redis ping fails - 503 + zoekt error when the gRPC call errors - 503 + redis error when Redis returns a non-PONG response - all three checks run in parallel (Promise.all), not serially The zoekt client is mocked via the new @/lib/zoektClient module so the test does not pull in @grpc/grpc-js at test-load time. --- .../api/(server)/health/ready/route.test.ts | 152 ++++++++++++++++++ 1 file changed, 152 insertions(+) create mode 100644 packages/web/src/app/api/(server)/health/ready/route.test.ts diff --git a/packages/web/src/app/api/(server)/health/ready/route.test.ts b/packages/web/src/app/api/(server)/health/ready/route.test.ts new file mode 100644 index 000000000..0a2972f26 --- /dev/null +++ b/packages/web/src/app/api/(server)/health/ready/route.test.ts @@ -0,0 +1,152 @@ +import { beforeEach, describe, expect, test, vi } from 'vitest'; + +const mocks = vi.hoisted(() => ({ + unsafePrisma: { + $queryRaw: vi.fn(), + }, + redisPing: vi.fn(), + zoektList: vi.fn(), +})); + +vi.mock('server-only', () => ({})); + +vi.mock('@/prisma', () => ({ + __unsafePrisma: mocks.unsafePrisma, +})); + +vi.mock('@/lib/redis', () => ({ + getRedisClient: () => ({ + ping: mocks.redisPing, + }), +})); + +vi.mock('@/lib/posthog', () => ({ + captureEvent: vi.fn(), +})); + +vi.mock('@/lib/zoektClient', () => ({ + loadZoektClient: () => ({ + List: mocks.zoektList, + }), +})); + +vi.mock('@sourcebot/shared', () => ({ + createLogger: () => ({ + debug: vi.fn(), + info: vi.fn(), + warn: vi.fn(), + error: vi.fn(), + }), +})); + +const { GET } = await import('./route'); + +describe('GET /api/health/ready', () => { + beforeEach(() => { + vi.clearAllMocks(); + mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]); + mocks.redisPing.mockResolvedValue('PONG'); + mocks.zoektList.mockImplementation( + (_request: unknown, callback: (err: Error | null) => void) => { + callback(null); + }, + ); + }); + + test('returns 200 with status:ok when all three dependencies are reachable', async () => { + const response = await GET(); + const body = await response.json(); + + expect(response.status).toBe(200); + expect(body.status).toBe('ok'); + expect(body.checks.postgres.status).toBe('ok'); + expect(body.checks.redis.status).toBe('ok'); + expect(body.checks.zoekt.status).toBe('ok'); + expect(typeof body.checks.postgres.latencyMs).toBe('number'); + expect(typeof body.checks.redis.latencyMs).toBe('number'); + expect(typeof body.checks.zoekt.latencyMs).toBe('number'); + }); + + test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => { + mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused')); + + const response = await GET(); + const body = await response.json(); + + expect(response.status).toBe(503); + expect(body.status).toBe('degraded'); + expect(body.checks.postgres.status).toBe('error'); + expect(body.checks.postgres.error).toBe('connection refused'); + expect(body.checks.redis.status).toBe('ok'); + expect(body.checks.zoekt.status).toBe('ok'); + }); + + test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => { + mocks.redisPing.mockRejectedValue(new Error('redis down')); + + const response = await GET(); + const body = await response.json(); + + expect(response.status).toBe(503); + expect(body.status).toBe('degraded'); + expect(body.checks.postgres.status).toBe('ok'); + expect(body.checks.redis.status).toBe('error'); + expect(body.checks.redis.error).toBe('redis down'); + expect(body.checks.zoekt.status).toBe('ok'); + }); + + test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => { + mocks.zoektList.mockImplementation( + (_request: unknown, callback: (err: Error | null) => void) => { + callback(new Error('UNAVAILABLE: zoekt not reachable')); + }, + ); + + const response = await GET(); + const body = await response.json(); + + expect(response.status).toBe(503); + expect(body.status).toBe('degraded'); + expect(body.checks.zoekt.status).toBe('error'); + expect(body.checks.zoekt.error).toContain('UNAVAILABLE'); + expect(body.checks.postgres.status).toBe('ok'); + expect(body.checks.redis.status).toBe('ok'); + }); + + test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => { + mocks.redisPing.mockResolvedValue('NOT-PONG'); + + const response = await GET(); + const body = await response.json(); + + expect(response.status).toBe(503); + expect(body.status).toBe('degraded'); + expect(body.checks.redis.status).toBe('error'); + expect(body.checks.redis.error).toContain('unexpected ping response'); + }); + + test('runs all three checks in parallel (Promise.all)', async () => { + const delay = 50; + mocks.unsafePrisma.$queryRaw.mockImplementation( + () => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)), + ); + mocks.redisPing.mockImplementation( + () => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)), + ); + mocks.zoektList.mockImplementation( + (_request: unknown, callback: (err: Error | null) => void) => { + setTimeout(() => callback(null), delay); + }, + ); + + const start = Date.now(); + const response = await GET(); + const elapsed = Date.now() - start; + const body = await response.json(); + + expect(response.status).toBe(200); + expect(body.status).toBe('ok'); + // Generous upper bound to avoid flakes; serial would be ~3x delay. + expect(elapsed).toBeLessThan(delay * 2.5); + }); +}); From e29ed6b3016e1c93105f58d28b7083f8116c822a Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Fri, 24 Jul 2026 23:10:05 +0530 Subject: [PATCH 04/10] docs(health): document liveness vs readiness split New docs/docs/api-reference/health.mdx describes both endpoints, the response shape, the per-dependency checks, and example Docker Compose and Kubernetes probes. Wired into the existing System group in docs/docs.json. --- docs/docs.json | 3 +- docs/docs/api-reference/health.mdx | 97 ++++++++++++++++++++++++++++++ 2 files changed, 99 insertions(+), 1 deletion(-) create mode 100644 docs/docs/api-reference/health.mdx diff --git a/docs/docs.json b/docs/docs.json index ad45e7fc8..ae26d3c8d 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -217,7 +217,8 @@ "icon": "server", "pages": [ "GET /api/version", - "GET /api/health" + "GET /api/health", + "docs/api-reference/health" ] } ] diff --git a/docs/docs/api-reference/health.mdx b/docs/docs/api-reference/health.mdx new file mode 100644 index 000000000..ec85f2c6a --- /dev/null +++ b/docs/docs/api-reference/health.mdx @@ -0,0 +1,97 @@ +--- +title: "Health Endpoints" +description: "Liveness and readiness probes for orchestrators and monitoring systems." +--- + +Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data. + +## Liveness: `GET /api/health` + +Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted. + +```bash +curl -fsS https://sourcebot.example.com/api/health +# {"status":"ok"} +``` + +## Readiness: `GET /api/health/ready` + +Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs. + +Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted. + +### Response shape + +```json +{ + "status": "ok", + "checks": { + "postgres": { "status": "ok", "latencyMs": 3 }, + "redis": { "status": "ok", "latencyMs": 1 }, + "zoekt": { "status": "ok", "latencyMs": 12 } + } +} +``` + +When degraded, each failed check carries an `error` field with the underlying message: + +```json +{ + "status": "degraded", + "checks": { + "postgres": { "status": "ok", "latencyMs": 4 }, + "redis": { "status": "ok", "latencyMs": 1 }, + "zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" } + } +} +``` + +| Check | What it probes | +|-------|---------------| +| `postgres` | `SELECT 1` via Prisma | +| `redis` | `PING` (rejects non-`PONG` responses) | +| `zoekt` | `List` RPC with a 1-second wall-time cap (proves the gRPC channel is alive) | + +### Example probes + + + + ```yaml + services: + sourcebot: + image: sourcebot/sourcebot:latest + healthcheck: + test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"] + interval: 30s + timeout: 5s + retries: 3 + # For dependency-aware probes, point the orchestrator at /api/health/ready + # instead. Sourcebot's example compose file does this via a sidecar. + ``` + + + ```yaml + livenessProbe: + httpGet: + path: /api/health + port: 3000 + initialDelaySeconds: 30 + periodSeconds: 30 + timeoutSeconds: 5 + failureThreshold: 3 + readinessProbe: + httpGet: + path: /api/health/ready + port: 3000 + initialDelaySeconds: 10 + periodSeconds: 10 + timeoutSeconds: 5 + successThreshold: 1 + failureThreshold: 3 + ``` + + + + +The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point. + From 375f2dd7b76d82b67d0935fe96103c24112d6b7b Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Fri, 24 Jul 2026 23:10:05 +0530 Subject: [PATCH 05/10] chore: changelog entry for /api/health/ready References issue #1506. --- CHANGELOG.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 92fdc34be..34214ab3e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added +- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1506](https://github.com/sourcebot-dev/sourcebot/issues/1506) + ### Changed - Vulnerability triage now keeps Linear issues synchronized with current security findings. From 2a47d1184a33bea42d50b5bf163da31d6683517b Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Sun, 26 Jul 2026 11:27:16 +0530 Subject: [PATCH 06/10] fix(web): only cache successful Zoekt client builds The previous implementation cached the first init attempt in zoektClientPromise and never cleared it. A transient startup error (proto load failure, network timeout) would then permanently mark the Zoekt readiness check as failed until the process restarted. This was a pre-existing latent bug in the zoektClient helper, but the readiness route made it visible. The fix: store the resolved client (not the in-flight promise) so a failed build never reaches the cache, and a subsequent call can retry the init. Addresses Bugbot finding cd93d60f on the same file. --- packages/web/src/lib/zoektClient.ts | 80 ++++++++++++++++------------- 1 file changed, 44 insertions(+), 36 deletions(-) diff --git a/packages/web/src/lib/zoektClient.ts b/packages/web/src/lib/zoektClient.ts index 1b02a6567..accaf3963 100644 --- a/packages/web/src/lib/zoektClient.ts +++ b/packages/web/src/lib/zoektClient.ts @@ -10,41 +10,49 @@ export type ZoektClient = { List: (request: ZoektListRequest, callback: (err: Error | null) => void) => void; }; -let zoektClientPromise: Promise | undefined; - -export const loadZoektClient = (): Promise => { - if (!zoektClientPromise) { - zoektClientPromise = (async () => { - const [grpc, protoLoader, nodePath, shared] = await Promise.all([ - import('@grpc/grpc-js'), - import('@grpc/proto-loader'), - import('node:path'), - import('@sourcebot/shared'), - ]); - - const protoBasePath = nodePath.join(process.cwd(), '../../vendor/zoekt/grpc/protos'); - const protoPath = nodePath.join(protoBasePath, 'zoekt/webserver/v1/webserver.proto'); - - const packageDefinition = protoLoader.loadSync(protoPath, { - keepCase: true, - longs: Number, - enums: String, - defaults: true, - oneofs: true, - includeDirs: [protoBasePath], - }); - - const proto = grpc.loadPackageDefinition(packageDefinition) as unknown as { - zoekt: { webserver: { v1: { WebserverService: new (address: string, credentials: unknown) => ZoektClient } } }; - }; - - const zoektUrl = new URL(shared.env.ZOEKT_WEBSERVER_URL); - const grpcAddress = `${zoektUrl.hostname}:${zoektUrl.port}`; - return new proto.zoekt.webserver.v1.WebserverService( - grpcAddress, - grpc.credentials.createInsecure(), - ); - })(); +let cachedClient: ZoektClient | undefined; + +const buildClient = async (): Promise => { + const [grpc, protoLoader, nodePath, shared] = await Promise.all([ + import('@grpc/grpc-js'), + import('@grpc/proto-loader'), + import('node:path'), + import('@sourcebot/shared'), + ]); + + const protoBasePath = nodePath.join(process.cwd(), '../../vendor/zoekt/grpc/protos'); + const protoPath = nodePath.join(protoBasePath, 'zoekt/webserver/v1/webserver.proto'); + + const packageDefinition = protoLoader.loadSync(protoPath, { + keepCase: true, + longs: Number, + enums: String, + defaults: true, + oneofs: true, + includeDirs: [protoBasePath], + }); + + const proto = grpc.loadPackageDefinition(packageDefinition) as unknown as { + zoekt: { webserver: { v1: { WebserverService: new (address: string, credentials: unknown) => ZoektClient } } }; + }; + + const zoektUrl = new URL(shared.env.ZOEKT_WEBSERVER_URL); + const grpcAddress = `${zoektUrl.hostname}:${zoektUrl.port}`; + return new proto.zoekt.webserver.v1.WebserverService( + grpcAddress, + grpc.credentials.createInsecure(), + ); +}; + +// Returns a connected client, building it on first use. Only successful +// builds are cached: a transient init failure does not poison subsequent +// readiness probes, so the next call can retry. +export const loadZoektClient = async (): Promise => { + if (cachedClient) { + return cachedClient; } - return zoektClientPromise; + const client = await buildClient(); + cachedClient = client; + return client; }; + From 7a4ddc53440e53011ebc82b87b2891b9eaee2a4d Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Sun, 26 Jul 2026 11:27:16 +0530 Subject: [PATCH 07/10] fix(web): bound the readiness route by its own timeout Two related issues surfaced from bot review on the readiness route: 1. `checkZoekt` called `loadZoektClient()` outside the `withTimeout` wrapper, so a stalled first-call Zoekt init (vendored proto load, network DNS, etc.) could exceed the documented 2s bound. Move the init call inside the timeout so it shares the same bound as the gRPC call. 2. `withTimeout` raced the check promise against the timeout and discarded the loser. If the loser later rejected, no one was awaiting it, surfacing as an unhandled-promise-rejection warning in the Node process during a hung-dependency outage. Attach a no-op `.catch` to the check promise so late rejections are absorbed; the visible result (the timeout error or the actual check error) is unchanged. Addresses Bugbot findings 597dbe01 and 367ae57f, and CodeRabbit finding 'Cancel timed-out readiness dependency operations' on the same file. --- .../src/app/api/(server)/health/ready/route.ts | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/packages/web/src/app/api/(server)/health/ready/route.ts b/packages/web/src/app/api/(server)/health/ready/route.ts index 2cadcbe3a..7b0bedd67 100644 --- a/packages/web/src/app/api/(server)/health/ready/route.ts +++ b/packages/web/src/app/api/(server)/health/ready/route.ts @@ -26,14 +26,19 @@ type ReadinessResponse = { }; }; -// Wraps a check function in a per-check timeout. When the timeout fires -// first, the check resolves as an error result; the underlying promise is -// allowed to settle in the background (its result is discarded). +// Runs `check()` and rejects if it has not settled after `timeoutMs`. When the +// timeout fires, the underlying check promise may still resolve or reject +// later; the no-op `.catch` below attaches to that promise so a late +// rejection does not surface as an unhandled-promise-rejection in the Node +// process while the readiness request has already moved on. const withTimeout = async ( label: string, check: () => Promise, timeoutMs: number, ): Promise => { + const checkPromise = check(); + checkPromise.catch(() => { /* swallowed: see comment above */ }); + let timer: ReturnType | undefined; const timeout = new Promise((_, reject) => { timer = setTimeout(() => { @@ -41,7 +46,7 @@ const withTimeout = async ( }, timeoutMs); }); try { - return await Promise.race([check(), timeout]); + return await Promise.race([checkPromise, timeout]); } finally { if (timer) { clearTimeout(timer); @@ -88,8 +93,10 @@ const checkRedis = async (): Promise => { const checkZoekt = async (): Promise => { const start = Date.now(); try { - const client = await loadZoektClient(); await withTimeout('zoekt', async () => { + // Build the client inside the timeout so a first-call init stall + // (e.g., vendored proto load) is also bounded. + const client = await loadZoektClient(); await new Promise((resolve, reject) => { // An empty List with a 1s wall-time cap is the smallest request // that exercises the gRPC channel end-to-end. It returns an From 56e336b6fb932ba7256ef8cb88f644466525df63 Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Sun, 26 Jul 2026 11:27:16 +0530 Subject: [PATCH 08/10] test(web): cover unhandled-rejection suppression in readiness probe New test attaches a process-level 'unhandledRejection' listener and asserts that the readiness probe does not surface the check promise's rejection to it, even when the race has already returned the timeout error. Locks in the no-op `.catch` in `withTimeout`. --- .../api/(server)/health/ready/route.test.ts | 33 +++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/packages/web/src/app/api/(server)/health/ready/route.test.ts b/packages/web/src/app/api/(server)/health/ready/route.test.ts index 0a2972f26..315b7539d 100644 --- a/packages/web/src/app/api/(server)/health/ready/route.test.ts +++ b/packages/web/src/app/api/(server)/health/ready/route.test.ts @@ -149,4 +149,37 @@ describe('GET /api/health/ready', () => { // Generous upper bound to avoid flakes; serial would be ~3x delay. expect(elapsed).toBeLessThan(delay * 2.5); }); + + test('does not surface check rejections as unhandled promise rejections', async () => { + // The check rejects synchronously (well within the 2s timeout). The + // no-op `.catch` attached in `withTimeout` must absorb that + // rejection so the Node process does not log an + // unhandled-promise-rejection warning while the readiness request + // has already moved on. + const checkRejection = new Error('check rejected'); + const unhandled: unknown[] = []; + const onUnhandled = (err: unknown) => { unhandled.push(err); }; + process.on('unhandledRejection', onUnhandled); + + try { + mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection); + mocks.redisPing.mockResolvedValue('PONG'); + mocks.zoektList.mockImplementation( + (_request: unknown, callback: (err: Error | null) => void) => { + callback(null); + }, + ); + + const response = await GET(); + const body = await response.json(); + + expect(response.status).toBe(503); + expect(body.checks.postgres.status).toBe('error'); + // Give the rejection microtask a chance to fire and propagate. + await new Promise((resolve) => setTimeout(resolve, 50)); + expect(unhandled).not.toContain(checkRejection); + } finally { + process.off('unhandledRejection', onUnhandled); + } + }); }); From 3b1ebe9ec5edddbf683d727b9dfb2bddfd5004b6 Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Sun, 26 Jul 2026 11:27:16 +0530 Subject: [PATCH 09/10] chore: changelog link to PR #1507 instead of issue #1506 Coding guidelines: changelog entries must reference the PR id, not the issue. The previous link pointed at the issue (which is what was known at the time the entry was written). This aligns the entry with the convention. --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 34214ab3e..78bcd0988 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,7 +8,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] ### Added -- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1506](https://github.com/sourcebot-dev/sourcebot/issues/1506) +- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507) ### Changed - Vulnerability triage now keeps Linear issues synchronized with current security findings. From f5193aeae5e7b6ef436f0a3ed39c57935b5d22b3 Mon Sep 17 00:00:00 2001 From: Harsh23Kashyap <55448981+Harsh23Kashyap@users.noreply.github.com> Date: Sun, 26 Jul 2026 11:34:43 +0530 Subject: [PATCH 10/10] fix(web): Zoekt readiness probe uses empty List options (max_wall_time is a SearchOptions field) Address Cursor Bugbot finding on the readiness probe (PR #1507 review #3): the gRPC `List` call was being issued with `{ opts: { max_wall_time: ... } }`, but `max_wall_time` is a `SearchOptions` field, not a `ListOptions` field. The Zoekt server silently dropped it, so the intended 1-second server-side timeout never applied and the call was only bounded by the 2-second client-side timeout in `READINESS_TIMEOUT_MS`. On a hung Zoekt instance the request would still return inside 2s, but the in-flight RPC could keep the Zoekt worker busy for longer than necessary. Drop the bogus `opts`, document why, and add a regression test that locks in `List` being called with an empty options object. The probe now issues the smallest valid request: `client.List({}, cb)`. --- docs/docs/api-reference/health.mdx | 2 +- .../api/(server)/health/ready/route.test.ts | 14 ++++++++++ .../app/api/(server)/health/ready/route.ts | 27 ++++++++++--------- 3 files changed, 29 insertions(+), 14 deletions(-) diff --git a/docs/docs/api-reference/health.mdx b/docs/docs/api-reference/health.mdx index ec85f2c6a..3550dd697 100644 --- a/docs/docs/api-reference/health.mdx +++ b/docs/docs/api-reference/health.mdx @@ -50,7 +50,7 @@ When degraded, each failed check carries an `error` field with the underlying me |-------|---------------| | `postgres` | `SELECT 1` via Prisma | | `redis` | `PING` (rejects non-`PONG` responses) | -| `zoekt` | `List` RPC with a 1-second wall-time cap (proves the gRPC channel is alive) | +| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) | ### Example probes diff --git a/packages/web/src/app/api/(server)/health/ready/route.test.ts b/packages/web/src/app/api/(server)/health/ready/route.test.ts index 315b7539d..a7203af01 100644 --- a/packages/web/src/app/api/(server)/health/ready/route.test.ts +++ b/packages/web/src/app/api/(server)/health/ready/route.test.ts @@ -182,4 +182,18 @@ describe('GET /api/health/ready', () => { process.off('unhandledRejection', onUnhandled); } }); + + test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => { + // Regression guard: the earlier draft of the Zoekt probe passed + // `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field + // belongs to `SearchOptions` and is silently ignored by `List` + // (whose `ListOptions` only carries `field`). The 2s client-side + // timeout is the only thing that actually bounds the call. The + // probe must therefore issue the smallest valid request, which is + // an empty options object. + const response = await GET(); + expect(response.status).toBe(200); + expect(mocks.zoektList).toHaveBeenCalledTimes(1); + expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function)); + }); }); diff --git a/packages/web/src/app/api/(server)/health/ready/route.ts b/packages/web/src/app/api/(server)/health/ready/route.ts index 7b0bedd67..3bb072e13 100644 --- a/packages/web/src/app/api/(server)/health/ready/route.ts +++ b/packages/web/src/app/api/(server)/health/ready/route.ts @@ -98,19 +98,20 @@ const checkZoekt = async (): Promise => { // (e.g., vendored proto load) is also bounded. const client = await loadZoektClient(); await new Promise((resolve, reject) => { - // An empty List with a 1s wall-time cap is the smallest request - // that exercises the gRPC channel end-to-end. It returns an - // empty result, not an error, even when no repos are indexed. - client.List( - { opts: { max_wall_time: { seconds: 1, nanos: 0 } } }, - (err) => { - if (err) { - reject(err); - } else { - resolve(); - } - }, - ); + // Empty `List` is the smallest request that exercises the gRPC + // channel end-to-end. It returns an empty result, not an error, + // even when no repos are indexed. The 2s client-side + // `READINESS_TIMEOUT_MS` is the only timeout that actually + // bounds the call: `max_wall_time` is a `SearchOptions` field + // and does not apply to `List` (`ListOptions` only carries + // `field`). + client.List({}, (err) => { + if (err) { + reject(err); + } else { + resolve(); + } + }); }); }, READINESS_TIMEOUT_MS); return { status: 'ok', latencyMs: Date.now() - start };