Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
199 changes: 199 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});

test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
// Regression guard: the earlier draft of the Zoekt probe passed
// `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
// belongs to `SearchOptions` and is silently ignored by `List`
// (whose `ListOptions` only carries `field`). The 2s client-side
// timeout is the only thing that actually bounds the call. The
// probe must therefore issue the smallest valid request, which is
// an empty options object.
const response = await GET();
expect(response.status).toBe(200);
expect(mocks.zoektList).toHaveBeenCalledTimes(1);
expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
});
});
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
199 changes: 199 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});

test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
// Regression guard: the earlier draft of the Zoekt probe passed
// `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
// belongs to `SearchOptions` and is silently ignored by `List`
// (whose `ListOptions` only carries `field`). The 2s client-side
// timeout is the only thing that actually bounds the call. The
// probe must therefore issue the smallest valid request, which is
// an empty options object.
const response = await GET();
expect(response.status).toBe(200);
expect(mocks.zoektList).toHaveBeenCalledTimes(1);
expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
});
});
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
199 changes: 199 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});

test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
// Regression guard: the earlier draft of the Zoekt probe passed
// `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
// belongs to `SearchOptions` and is silently ignored by `List`
// (whose `ListOptions` only carries `field`). The 2s client-side
// timeout is the only thing that actually bounds the call. The
// probe must therefore issue the smallest valid request, which is
// an empty options object.
const response = await GET();
expect(response.status).toBe(200);
expect(mocks.zoektList).toHaveBeenCalledTimes(1);
expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
});
});
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
199 changes: 199 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});

test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
// Regression guard: the earlier draft of the Zoekt probe passed
// `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
// belongs to `SearchOptions` and is silently ignored by `List`
// (whose `ListOptions` only carries `field`). The 2s client-side
// timeout is the only thing that actually bounds the call. The
// probe must therefore issue the smallest valid request, which is
// an empty options object.
const response = await GET();
expect(response.status).toBe(200);
expect(mocks.zoektList).toHaveBeenCalledTimes(1);
expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
});
});
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
199 changes: 199 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});

test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
// Regression guard: the earlier draft of the Zoekt probe passed
// `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
// belongs to `SearchOptions` and is silently ignored by `List`
// (whose `ListOptions` only carries `field`). The 2s client-side
// timeout is the only thing that actually bounds the call. The
// probe must therefore issue the smallest valid request, which is
// an empty options object.
const response = await GET();
expect(response.status).toBe(200);
expect(mocks.zoektList).toHaveBeenCalledTimes(1);
expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
});
});
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
199 changes: 199 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});

test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
// Regression guard: the earlier draft of the Zoekt probe passed
// `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
// belongs to `SearchOptions` and is silently ignored by `List`
// (whose `ListOptions` only carries `field`). The 2s client-side
// timeout is the only thing that actually bounds the call. The
// probe must therefore issue the smallest valid request, which is
// an empty options object.
const response = await GET();
expect(response.status).toBe(200);
expect(mocks.zoektList).toHaveBeenCalledTimes(1);
expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
});
});
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
199 changes: 199 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});

test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
// Regression guard: the earlier draft of the Zoekt probe passed
// `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
// belongs to `SearchOptions` and is silently ignored by `List`
// (whose `ListOptions` only carries `field`). The 2s client-side
// timeout is the only thing that actually bounds the call. The
// probe must therefore issue the smallest valid request, which is
// an empty options object.
const response = await GET();
expect(response.status).toBe(200);
expect(mocks.zoektList).toHaveBeenCalledTimes(1);
expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
});
});
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added a `GET /api/health/ready` endpoint that returns per-dependency health (Postgres, Redis, Zoekt) for use as a Kubernetes `readinessProbe` or load-balancer health check. The existing `GET /api/health` endpoint is unchanged and remains the liveness probe. [#1507](https://github.com/sourcebot-dev/sourcebot/pull/1507)

### Changed
- Vulnerability triage now keeps Linear issues synchronized with current security findings.

Expand Down
3 changes: 2 additions & 1 deletion docs/docs.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -217,7 +217,8 @@
"icon": "server",
"pages": [
"GET /api/version",
"GET /api/health"
"GET /api/health",
"docs/api-reference/health"
]
}
]
Expand Down
97 changes: 97 additions & 0 deletions docs/docs/api-reference/health.mdx
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
---
title: "Health Endpoints"
description: "Liveness and readiness probes for orchestrators and monitoring systems."
---

Sourcebot exposes two public health endpoints that follow the standard Kubernetes liveness / readiness split. Both are unauthenticated and return no user data.

## Liveness: `GET /api/health`

Returns `200 OK` with `{ "status": "ok" }` whenever the Next.js process is running and able to handle a request. Does not touch the database, Redis, or Zoekt. Use this for Kubernetes `livenessProbe` or Docker Compose `healthcheck.test`. A failing liveness probe means the process must be restarted.

```bash
curl -fsS https://sourcebot.example.com/api/health
# {"status":"ok"}
```

## Readiness: `GET /api/health/ready`

Returns `200 OK` with `{"status":"ok", "checks":{...}}` when Postgres, Redis, and Zoekt are all reachable. Returns `503 Service Unavailable` with `{"status":"degraded", "checks":{...}}` if any dependency is unreachable. Each check runs in parallel with a 2-second per-check timeout, so the worst-case request time is bounded even when a dependency hangs.

Use this for Kubernetes `readinessProbe` or a load balancer health check. A failing readiness probe means the pod should be removed from the load-balancer rotation but not restarted.

### Response shape

```json
{
"status": "ok",
"checks": {
"postgres": { "status": "ok", "latencyMs": 3 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "ok", "latencyMs": 12 }
}
}
```

When degraded, each failed check carries an `error` field with the underlying message:

```json
{
"status": "degraded",
"checks": {
"postgres": { "status": "ok", "latencyMs": 4 },
"redis": { "status": "ok", "latencyMs": 1 },
"zoekt": { "status": "error", "latencyMs": 2003, "error": "zoekt check timed out after 2000ms" }
}
}
```

| Check | What it probes |
|-------|---------------|
| `postgres` | `SELECT 1` via Prisma |
| `redis` | `PING` (rejects non-`PONG` responses) |
| `zoekt` | Empty `List` RPC (proves the gRPC channel is alive; bounded by the 2s per-check timeout) |

### Example probes

<Tabs>
<Tab title="Docker Compose">
```yaml
services:
sourcebot:
image: sourcebot/sourcebot:latest
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:3000/api/health"]
interval: 30s
timeout: 5s
retries: 3
# For dependency-aware probes, point the orchestrator at /api/health/ready
# instead. Sourcebot's example compose file does this via a sidecar.
```
</Tab>
<Tab title="Kubernetes">
```yaml
livenessProbe:
httpGet:
path: /api/health
port: 3000
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
readinessProbe:
httpGet:
path: /api/health/ready
port: 3000
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
```
</Tab>
</Tabs>

<Note>
The readiness probe hits the database on every call. On large deployments with many pods, a high-frequency probe interval (sub-5s) can produce noticeable background load. A 10s interval with `failureThreshold: 3` is a good starting point.
</Note>
199 changes: 199 additions & 0 deletions packages/web/src/app/api/(server)/health/ready/route.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
import { beforeEach, describe, expect, test, vi } from 'vitest';

const mocks = vi.hoisted(() => ({
unsafePrisma: {
$queryRaw: vi.fn(),
},
redisPing: vi.fn(),
zoektList: vi.fn(),
}));

vi.mock('server-only', () => ({}));

vi.mock('@/prisma', () => ({
__unsafePrisma: mocks.unsafePrisma,
}));

vi.mock('@/lib/redis', () => ({
getRedisClient: () => ({
ping: mocks.redisPing,
}),
}));

vi.mock('@/lib/posthog', () => ({
captureEvent: vi.fn(),
}));

vi.mock('@/lib/zoektClient', () => ({
loadZoektClient: () => ({
List: mocks.zoektList,
}),
}));

vi.mock('@sourcebot/shared', () => ({
createLogger: () => ({
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
}),
}));

const { GET } = await import('./route');

describe('GET /api/health/ready', () => {
beforeEach(() => {
vi.clearAllMocks();
mocks.unsafePrisma.$queryRaw.mockResolvedValue([{ '?column?': 1 }]);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);
});

test('returns 200 with status:ok when all three dependencies are reachable', async () => {
const response = await GET();
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
expect(typeof body.checks.postgres.latencyMs).toBe('number');
expect(typeof body.checks.redis.latencyMs).toBe('number');
expect(typeof body.checks.zoekt.latencyMs).toBe('number');
});

test('returns 503 with status:degraded and a postgres error when Postgres is unreachable', async () => {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(new Error('connection refused'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('error');
expect(body.checks.postgres.error).toBe('connection refused');
expect(body.checks.redis.status).toBe('ok');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded and a redis error when Redis ping fails', async () => {
mocks.redisPing.mockRejectedValue(new Error('redis down'));

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toBe('redis down');
expect(body.checks.zoekt.status).toBe('ok');
});

test('returns 503 with status:degraded when the Zoekt gRPC call errors', async () => {
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(new Error('UNAVAILABLE: zoekt not reachable'));
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.zoekt.status).toBe('error');
expect(body.checks.zoekt.error).toContain('UNAVAILABLE');
expect(body.checks.postgres.status).toBe('ok');
expect(body.checks.redis.status).toBe('ok');
});

test('returns 503 with status:degraded when Redis returns a non-PONG response', async () => {
mocks.redisPing.mockResolvedValue('NOT-PONG');

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.status).toBe('degraded');
expect(body.checks.redis.status).toBe('error');
expect(body.checks.redis.error).toContain('unexpected ping response');
});

test('runs all three checks in parallel (Promise.all)', async () => {
const delay = 50;
mocks.unsafePrisma.$queryRaw.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve([{}]), delay)),
);
mocks.redisPing.mockImplementation(
() => new Promise((resolve) => setTimeout(() => resolve('PONG'), delay)),
);
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
setTimeout(() => callback(null), delay);
},
);

const start = Date.now();
const response = await GET();
const elapsed = Date.now() - start;
const body = await response.json();

expect(response.status).toBe(200);
expect(body.status).toBe('ok');
// Generous upper bound to avoid flakes; serial would be ~3x delay.
expect(elapsed).toBeLessThan(delay * 2.5);
});

test('does not surface check rejections as unhandled promise rejections', async () => {
// The check rejects synchronously (well within the 2s timeout). The
// no-op `.catch` attached in `withTimeout` must absorb that
// rejection so the Node process does not log an
// unhandled-promise-rejection warning while the readiness request
// has already moved on.
const checkRejection = new Error('check rejected');
const unhandled: unknown[] = [];
const onUnhandled = (err: unknown) => { unhandled.push(err); };
process.on('unhandledRejection', onUnhandled);

try {
mocks.unsafePrisma.$queryRaw.mockRejectedValue(checkRejection);
mocks.redisPing.mockResolvedValue('PONG');
mocks.zoektList.mockImplementation(
(_request: unknown, callback: (err: Error | null) => void) => {
callback(null);
},
);

const response = await GET();
const body = await response.json();

expect(response.status).toBe(503);
expect(body.checks.postgres.status).toBe('error');
// Give the rejection microtask a chance to fire and propagate.
await new Promise((resolve) => setTimeout(resolve, 50));
expect(unhandled).not.toContain(checkRejection);
} finally {
process.off('unhandledRejection', onUnhandled);
}
});

test('issues the Zoekt List RPC with empty options (max_wall_time is a SearchOptions field, not ListOptions)', async () => {
// Regression guard: the earlier draft of the Zoekt probe passed
// `{ opts: { max_wall_time: ... } }` to the `List` RPC. That field
// belongs to `SearchOptions` and is silently ignored by `List`
// (whose `ListOptions` only carries `field`). The 2s client-side
// timeout is the only thing that actually bounds the call. The
// probe must therefore issue the smallest valid request, which is
// an empty options object.
const response = await GET();
expect(response.status).toBe(200);
expect(mocks.zoektList).toHaveBeenCalledTimes(1);
expect(mocks.zoektList).toHaveBeenCalledWith({}, expect.any(Function));
});
});
Loading